Add Dagster orchestration and reduce active brands in backend

This commit is contained in:
sriram
2026-08-20 16:39:54 +05:30
parent fbb1356e47
commit 7bf8dc6922
66 changed files with 2664 additions and 21 deletions

View File

@@ -0,0 +1,130 @@
"""The single source of truth for which brands the application works on.
WHY THIS EXISTS
---------------
The dev dataset grew to 31 seed catalogs and 16 `brand_*` tables. Every
"all brands" query fans out across every table, boot parses ~19MB of seed
JSON, and 14 of those seed files have no table yet - so the next auto-seed
would silently create 14 more. That is far more than an 8GB dev machine
needs to exercise the features.
Rather than delete data, ONE setting decides which brands participate:
ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
and every read path, pipeline, RAG query, MCP tool and Dagster asset
inherits it, because they all funnel through `vector_store`'s two brand
discovery functions (`list_available_brands` and `_list_brand_table_suffixes`)
plus `brand_sync.load_seed_catalogs`. Going back to 5, 10 or 25 brands is a
one-line `.env` change, not a code edit.
THE EMPTY-MEANS-ALL CONTRACT
----------------------------
An unset or blank `ACTIVE_BRANDS` disables filtering entirely, so this module
is a no-op on any deployment that does not opt in. That is deliberate:
`.env.production` leaves it unset, so production keeps serving every brand
while local development runs on three. It also means the feature can be
switched off wholesale if it ever gets in the way.
NAMES ARE RESOLVED, NOT MATCHED
-------------------------------
Configured names go through `resolve_parent_brand` -> `_sanitize_name`, the
same two steps that decide which table a product is stored in. So
`ACTIVE_BRANDS=Tata` activates `brand_hindustan_unilever` (the "hul tata tea"
alias claims it), exactly like an ingest of that brand would. Comparing raw
strings here would have let the config and the storage layer disagree.
"""
from __future__ import annotations
from typing import FrozenSet, Iterable, List, Optional
from app.infrastructure.settings import ACTIVE_BRANDS as _RAW_ACTIVE_BRANDS
from app.services.brand_registry import resolve_parent_brand
# Parsed once. `_CACHE` holds `(suffixes, display_names)`; `None` in slot 0
# means "no filtering configured", which is different from "an empty set of
# active brands" - the latter would hide the entire catalog.
_CACHE: Optional[tuple] = None
def _sanitize(name: str) -> str:
"""Brand name -> table suffix.
Imported lazily from vector_store because vector_store imports THIS module
at the top level; a top-level import here would be a cycle. query_intent
already breaks the identical cycle the same way.
"""
from app.services.vector_store import _sanitize_name
return _sanitize_name(name)
def _parse(raw: str) -> tuple:
names = [part.strip() for part in (raw or "").split(",")]
names = [n for n in names if n]
if not names:
return (None, [])
suffixes = []
display = []
for name in names:
suffix = _sanitize(resolve_parent_brand(name))
if not suffix or suffix in suffixes:
continue
suffixes.append(suffix)
display.append(name)
return (frozenset(suffixes), display)
def _load() -> tuple:
global _CACHE
if _CACHE is None:
_CACHE = _parse(_RAW_ACTIVE_BRANDS)
return _CACHE
def invalidate() -> None:
"""Drop the parsed config. For tests that monkeypatch the raw setting."""
global _CACHE
_CACHE = None
def filtering_enabled() -> bool:
"""True when ACTIVE_BRANDS is set to a non-empty list."""
return _load()[0] is not None
def active_brand_suffixes() -> Optional[FrozenSet[str]]:
"""Table suffixes of the active brands, or None when filtering is off."""
return _load()[0]
def is_active_suffix(suffix: str) -> bool:
"""Whether a `brand_<suffix>` table participates. True for all when off."""
active = _load()[0]
return True if active is None else (suffix or "").lower() in active
def filter_suffixes(suffixes: Iterable[str]) -> List[str]:
"""Keep only the active suffixes, preserving the caller's order."""
active = _load()[0]
if active is None:
return list(suffixes)
return [s for s in suffixes if (s or "").lower() in active]
def is_active_brand(brand: str) -> bool:
"""Whether a brand NAME (in any alias form) resolves to an active table."""
active = _load()[0]
if active is None:
return True
return _sanitize(resolve_parent_brand(brand)) in active
def active_display_names() -> List[str]:
"""The configured names, verbatim, for logs and Dagster partition keys.
Empty when filtering is off - callers that need the real brand list in
that case should ask `vector_store.list_available_brands()` instead.
"""
return list(_load()[1])

View File

@@ -29,6 +29,11 @@ from pathlib import Path
from typing import Any, Dict, List, Optional
from app.services.brand_registry import resolve_parent_brand
from app.services.active_brands import (
active_brand_suffixes,
is_active_brand,
is_active_suffix,
)
from app.services.vector_store import (
_connect,
_list_brand_table_suffixes,
@@ -45,6 +50,28 @@ logger = logging.getLogger(__name__)
# app/services/brand_sync.py -> parents[2] is backend/
SEED_DIR = Path(__file__).resolve().parents[2] / "data" / "seed_catalogs"
# Catalogs for brands outside ACTIVE_BRANDS live in this subdirectory. It is a
# plain subfolder rather than a separate tree so the two stay side by side, and
# it is NOT matched by SEED_DIR.glob("*.json") - which is what keeps the boot
# auto-seed from parsing ~19MB and creating a table for every archived brand.
ARCHIVE_DIR_NAME = "archive"
def seed_catalog_paths(seed_dir: Path = SEED_DIR) -> List[Path]:
"""Every seed catalog, active directory first, then the archive.
The archive is searched too, deliberately: re-activating a brand must be a
one-line ACTIVE_BRANDS change, so where a file physically sits is
presentation, not policy. Filtering by brand happens in load_seed_catalogs.
"""
if not seed_dir.exists():
return []
paths = sorted(seed_dir.glob("*.json"))
archive = seed_dir / ARCHIVE_DIR_NAME
if archive.is_dir():
paths.extend(sorted(archive.glob("*.json")))
return paths
# The column list written by upsert_brand_products, minus `embedding` (handled
# separately because it is huge and optional). Keeping these in sync is what
# makes an exported file round-trip back through the seeder without losing
@@ -95,7 +122,7 @@ def index_seed_files(seed_dir: Path = SEED_DIR) -> Dict[str, List[Path]]:
index: Dict[str, List[Path]] = defaultdict(list)
if not seed_dir.exists():
return {}
for path in sorted(seed_dir.glob("*.json")):
for path in seed_catalog_paths(seed_dir):
data = _read_catalog(path)
if data is None:
continue
@@ -267,25 +294,69 @@ def load_seed_catalogs(seed_dir: Path = SEED_DIR,
logger.error("Seed directory not found: %s", seed_dir)
return {}
files = sorted(seed_dir.glob("*.json"))
files = seed_catalog_paths(seed_dir)
if only:
wanted = [w.lower() for w in only]
files = [f for f in files if any(w in f.name.lower() for w in wanted)]
# `only` is an explicit request for named files, so it wins over the
# ACTIVE_BRANDS filter - that is how an archived brand can still be
# re-ingested deliberately without first editing the config.
respect_active = not only
brand_products: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
skipped_inactive = 0
for path in files:
data = _read_catalog(path)
if data is None:
logger.warning("Skipping %s - no brand/products found", path.name)
continue
resolved = resolve_parent_brand(data["brand"])
if respect_active and not is_active_brand(resolved):
skipped_inactive += 1
continue
brand_products[resolved].extend(data["products"])
logger.info("Read %d products from %s -> resolved brand '%s'",
len(data["products"]), path.name, resolved)
if skipped_inactive:
logger.info("Skipped %d seed catalog(s) outside ACTIVE_BRANDS", skipped_inactive)
return dict(brand_products)
def load_brand_products(brand: str) -> List[Dict[str, Any]]:
"""Every seed product belonging to `brand`, resolved properly.
Use this instead of `load_seed_catalogs(only=[brand])` when you have a
BRAND NAME. `only=` is a case-insensitive substring match on the FILE NAME,
which is the right thing for `seed_sample_data.py --only amul` but the
wrong thing for a brand:
* "Hindustan Unilever" contains a space; the file is
`brand_catalog_hindustan_unilever.json`. The substring never matches
and you silently get zero products - no error, just an empty result.
* Even with the slug, a filename match misses the other files that feed
the same table: `brand_catalog_tata.json` holds 121 Hindustan Unilever
products because the "hul tata tea" alias claims it.
Going through `index_seed_files()` fixes both: it is keyed by the resolved
table suffix and its value is the full list of files feeding that table.
Archived catalogs are included, so an explicitly requested brand is found
whether or not it is currently active.
"""
index = index_seed_files()
products: List[Dict[str, Any]] = []
for path in index.get(brand_slug(brand), []):
data = _read_catalog(path)
if data is None:
continue
products.extend(data["products"])
logger.info("Read %d products from %s for brand '%s'",
len(data["products"]), path.name, brand)
return products
def seed_brands(brand_products: Dict[str, List[Dict[str, Any]]], cleanup: bool = True) -> int:
"""Upsert grouped products into their brand tables. Returns the row count."""
total = 0
@@ -339,7 +410,7 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
by_slug: Dict[str, set] = defaultdict(set)
if not seed_dir.exists():
return []
for path in sorted(seed_dir.glob("*.json")):
for path in seed_catalog_paths(seed_dir):
data = _read_catalog(path)
if data is None:
continue
@@ -354,6 +425,9 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
"""Repair the table <-> seed-file correspondence in both directions.
Scoped to ACTIVE_BRANDS when that is set: an archived brand is neither
exported nor seeded, in either direction.
Idempotent and non-destructive:
* A populated table with no seed file gets one exported.
@@ -373,6 +447,22 @@ def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
file_index = index_seed_files()
collisions = _detect_collisions()
# BOTH SIDES MUST BE NARROWED TO THE ACTIVE BRANDS, OR NEITHER.
#
# _db_brand_counts() goes through _list_brand_table_suffixes(), so under
# ACTIVE_BRANDS it only sees the active tables. index_seed_files() reads the
# archive directory too, deliberately, so that re-activating a brand needs
# only a config change.
#
# Left mismatched, every archived brand looks like "a seed file whose table
# is empty" and lands in `to_seed` - so the boot reconcile would re-seed all
# 27 archived catalogs on the next restart, recreate their tables, and
# silently undo the archiving. Verified: a dry run reported exactly that.
if active_brand_suffixes() is not None:
file_index = {
slug: paths for slug, paths in file_index.items() if is_active_suffix(slug)
}
to_export = sorted(
suffix for suffix, count in db_counts.items()
if count > 0 and suffix not in file_index

View File

@@ -24,8 +24,25 @@ from app.services.discount_service import predict_discounts_for_store
logger = logging.getLogger(__name__)
# Every model this module knows how to train. Still the full set accepted by
# scripts/train_ml_models.py and POST /api/admin/store-intelligence/train.
ALL_MODELS = ["discount", "trending", "popularity", "forecast", "store_performance", "purchase_propensity"]
# What train_all() trains when the caller does not name a subset.
#
# `forecast`, `store_performance` and `purchase_propensity` are omitted
# deliberately: all three fit cleanly, but nothing reads their output back.
# store_db.get_latest_demand_forecast() has zero callers, and neither of the
# other two has an inference consumer anywhere in the app or the MCP tools.
# Training them by default spent roughly half of every retrain, and ~5MB of
# artifacts, producing bundles no request would ever load.
#
# Their training code, CLI flags and API options are untouched - ask for one by
# name (`--models store_performance`) and it trains and is written to
# MODEL_ARTIFACTS_DIR exactly as before. Move a model back into this list once
# something actually serves it.
PRODUCTION_MODELS = ["discount", "trending", "popularity"]
def _load_common_frames():
order_items = store_db.get_order_items_df()
@@ -144,7 +161,11 @@ def train_purchase_propensity(orders: pd.DataFrame) -> Dict:
def train_all(models: Optional[List[str]] = None) -> Dict[str, Dict]:
models = models or ALL_MODELS
"""Train `models`, defaulting to the models that are actually served.
Pass an explicit list (including any of ALL_MODELS) to train more.
"""
models = models or PRODUCTION_MODELS
order_items, orders, store_products = _load_common_frames()
results: Dict[str, Dict] = {}

View File

@@ -205,8 +205,16 @@ def _brand_index() -> tuple[list[str], Dict[str, str]]:
if _MENTION_CACHE["map"] is not None and now - _MENTION_CACHE["at"] < BRAND_MENTION_TTL_SECONDS:
return _MENTION_CACHE["known"], _MENTION_CACHE["map"]
known = list(KNOWN_BRANDS)
mapping = dict(BRAND_SEARCH_MAP)
# KNOWN_BRANDS/BRAND_SEARCH_MAP are a hand-tuned static table of 16 brands.
# Under ACTIVE_BRANDS they must be narrowed too, not just added to: without
# this a query for an archived brand ("Nestle") still parses as a brand
# mention, gets routed to brand_catalog mode, and returns zero rows - the
# same silent wrong-result failure mode as fuzzy category detection. Dropped
# here, it is treated as ordinary search text instead.
from app.services.active_brands import is_active_brand
known = [b for b in KNOWN_BRANDS if is_active_brand(b)]
mapping = {k: v for k, v in BRAND_SEARCH_MAP.items() if is_active_brand(v)}
try:
from app.services.vector_store import list_available_brands

View File

@@ -14,6 +14,7 @@ from app.infrastructure.settings import (
DB_CONNECT_TIMEOUT_SECONDS,
)
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.active_brands import filter_suffixes, is_active_suffix
from app.services.s3_service import s3_service
@@ -562,6 +563,12 @@ def list_available_brands() -> List[str]:
# Extract brand name from table name (e.g., brand_britannia -> britannia)
table_name = table[0]
suffix = table_name[len("brand_"):].lower() if table_name.startswith("brand_") else table_name.lower()
# ACTIVE_BRANDS narrows the whole app here: this list feeds
# /api/brands, the MCP list_brands tool, rag_service,
# query_intent's brand index, nutrition enrichment, store_db
# and get_products_all_brands. The table itself is untouched.
if not is_active_suffix(suffix):
continue
# Prefer the original display name from the brand map if available
brands.append(display_name_for_suffix(suffix, brand_map))
except Exception as e:
@@ -1220,12 +1227,23 @@ def _table_exists(cur, table_name: str) -> bool:
return bool(row and row[0])
def _list_brand_table_suffixes(cur) -> List[str]:
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table."""
def _list_brand_table_suffixes(cur, *, include_inactive: bool = False) -> List[str]:
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table.
Narrowed to ACTIVE_BRANDS by default. This is the shared chokepoint for
get_brand_overview, semantic_search, text_search, lexical_search,
list_all_categories and brand_sync._db_brand_counts, which is why the
active-brand setting reaches all of them without touching any of them.
`include_inactive=True` is the escape hatch for callers that address an
archived brand deliberately - an explicit re-ingest or a backfill - so
narrowing the catalog never means losing the ability to maintain it.
"""
cur.execute(
"""
SELECT table_name FROM information_schema.tables
WHERE table_schema = 'public' AND table_name LIKE 'brand_%'
"""
)
return [row[0][len("brand_"):] for row in cur.fetchall()]
suffixes = [row[0][len("brand_"):] for row in cur.fetchall()]
return suffixes if include_inactive else filter_suffixes(suffixes)