Add Dagster orchestration and reduce active brands in backend
This commit is contained in:
130
app/services/active_brands.py
Normal file
130
app/services/active_brands.py
Normal file
@@ -0,0 +1,130 @@
|
||||
"""The single source of truth for which brands the application works on.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
The dev dataset grew to 31 seed catalogs and 16 `brand_*` tables. Every
|
||||
"all brands" query fans out across every table, boot parses ~19MB of seed
|
||||
JSON, and 14 of those seed files have no table yet - so the next auto-seed
|
||||
would silently create 14 more. That is far more than an 8GB dev machine
|
||||
needs to exercise the features.
|
||||
|
||||
Rather than delete data, ONE setting decides which brands participate:
|
||||
|
||||
ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
|
||||
|
||||
and every read path, pipeline, RAG query, MCP tool and Dagster asset
|
||||
inherits it, because they all funnel through `vector_store`'s two brand
|
||||
discovery functions (`list_available_brands` and `_list_brand_table_suffixes`)
|
||||
plus `brand_sync.load_seed_catalogs`. Going back to 5, 10 or 25 brands is a
|
||||
one-line `.env` change, not a code edit.
|
||||
|
||||
THE EMPTY-MEANS-ALL CONTRACT
|
||||
----------------------------
|
||||
An unset or blank `ACTIVE_BRANDS` disables filtering entirely, so this module
|
||||
is a no-op on any deployment that does not opt in. That is deliberate:
|
||||
`.env.production` leaves it unset, so production keeps serving every brand
|
||||
while local development runs on three. It also means the feature can be
|
||||
switched off wholesale if it ever gets in the way.
|
||||
|
||||
NAMES ARE RESOLVED, NOT MATCHED
|
||||
-------------------------------
|
||||
Configured names go through `resolve_parent_brand` -> `_sanitize_name`, the
|
||||
same two steps that decide which table a product is stored in. So
|
||||
`ACTIVE_BRANDS=Tata` activates `brand_hindustan_unilever` (the "hul tata tea"
|
||||
alias claims it), exactly like an ingest of that brand would. Comparing raw
|
||||
strings here would have let the config and the storage layer disagree.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import FrozenSet, Iterable, List, Optional
|
||||
|
||||
from app.infrastructure.settings import ACTIVE_BRANDS as _RAW_ACTIVE_BRANDS
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
|
||||
# Parsed once. `_CACHE` holds `(suffixes, display_names)`; `None` in slot 0
|
||||
# means "no filtering configured", which is different from "an empty set of
|
||||
# active brands" - the latter would hide the entire catalog.
|
||||
_CACHE: Optional[tuple] = None
|
||||
|
||||
|
||||
def _sanitize(name: str) -> str:
|
||||
"""Brand name -> table suffix.
|
||||
|
||||
Imported lazily from vector_store because vector_store imports THIS module
|
||||
at the top level; a top-level import here would be a cycle. query_intent
|
||||
already breaks the identical cycle the same way.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
return _sanitize_name(name)
|
||||
|
||||
|
||||
def _parse(raw: str) -> tuple:
|
||||
names = [part.strip() for part in (raw or "").split(",")]
|
||||
names = [n for n in names if n]
|
||||
if not names:
|
||||
return (None, [])
|
||||
|
||||
suffixes = []
|
||||
display = []
|
||||
for name in names:
|
||||
suffix = _sanitize(resolve_parent_brand(name))
|
||||
if not suffix or suffix in suffixes:
|
||||
continue
|
||||
suffixes.append(suffix)
|
||||
display.append(name)
|
||||
return (frozenset(suffixes), display)
|
||||
|
||||
|
||||
def _load() -> tuple:
|
||||
global _CACHE
|
||||
if _CACHE is None:
|
||||
_CACHE = _parse(_RAW_ACTIVE_BRANDS)
|
||||
return _CACHE
|
||||
|
||||
|
||||
def invalidate() -> None:
|
||||
"""Drop the parsed config. For tests that monkeypatch the raw setting."""
|
||||
global _CACHE
|
||||
_CACHE = None
|
||||
|
||||
|
||||
def filtering_enabled() -> bool:
|
||||
"""True when ACTIVE_BRANDS is set to a non-empty list."""
|
||||
return _load()[0] is not None
|
||||
|
||||
|
||||
def active_brand_suffixes() -> Optional[FrozenSet[str]]:
|
||||
"""Table suffixes of the active brands, or None when filtering is off."""
|
||||
return _load()[0]
|
||||
|
||||
|
||||
def is_active_suffix(suffix: str) -> bool:
|
||||
"""Whether a `brand_<suffix>` table participates. True for all when off."""
|
||||
active = _load()[0]
|
||||
return True if active is None else (suffix or "").lower() in active
|
||||
|
||||
|
||||
def filter_suffixes(suffixes: Iterable[str]) -> List[str]:
|
||||
"""Keep only the active suffixes, preserving the caller's order."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return list(suffixes)
|
||||
return [s for s in suffixes if (s or "").lower() in active]
|
||||
|
||||
|
||||
def is_active_brand(brand: str) -> bool:
|
||||
"""Whether a brand NAME (in any alias form) resolves to an active table."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return True
|
||||
return _sanitize(resolve_parent_brand(brand)) in active
|
||||
|
||||
|
||||
def active_display_names() -> List[str]:
|
||||
"""The configured names, verbatim, for logs and Dagster partition keys.
|
||||
|
||||
Empty when filtering is off - callers that need the real brand list in
|
||||
that case should ask `vector_store.list_available_brands()` instead.
|
||||
"""
|
||||
return list(_load()[1])
|
||||
@@ -29,6 +29,11 @@ from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services.active_brands import (
|
||||
active_brand_suffixes,
|
||||
is_active_brand,
|
||||
is_active_suffix,
|
||||
)
|
||||
from app.services.vector_store import (
|
||||
_connect,
|
||||
_list_brand_table_suffixes,
|
||||
@@ -45,6 +50,28 @@ logger = logging.getLogger(__name__)
|
||||
# app/services/brand_sync.py -> parents[2] is backend/
|
||||
SEED_DIR = Path(__file__).resolve().parents[2] / "data" / "seed_catalogs"
|
||||
|
||||
# Catalogs for brands outside ACTIVE_BRANDS live in this subdirectory. It is a
|
||||
# plain subfolder rather than a separate tree so the two stay side by side, and
|
||||
# it is NOT matched by SEED_DIR.glob("*.json") - which is what keeps the boot
|
||||
# auto-seed from parsing ~19MB and creating a table for every archived brand.
|
||||
ARCHIVE_DIR_NAME = "archive"
|
||||
|
||||
|
||||
def seed_catalog_paths(seed_dir: Path = SEED_DIR) -> List[Path]:
|
||||
"""Every seed catalog, active directory first, then the archive.
|
||||
|
||||
The archive is searched too, deliberately: re-activating a brand must be a
|
||||
one-line ACTIVE_BRANDS change, so where a file physically sits is
|
||||
presentation, not policy. Filtering by brand happens in load_seed_catalogs.
|
||||
"""
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
paths = sorted(seed_dir.glob("*.json"))
|
||||
archive = seed_dir / ARCHIVE_DIR_NAME
|
||||
if archive.is_dir():
|
||||
paths.extend(sorted(archive.glob("*.json")))
|
||||
return paths
|
||||
|
||||
# The column list written by upsert_brand_products, minus `embedding` (handled
|
||||
# separately because it is huge and optional). Keeping these in sync is what
|
||||
# makes an exported file round-trip back through the seeder without losing
|
||||
@@ -95,7 +122,7 @@ def index_seed_files(seed_dir: Path = SEED_DIR) -> Dict[str, List[Path]]:
|
||||
index: Dict[str, List[Path]] = defaultdict(list)
|
||||
if not seed_dir.exists():
|
||||
return {}
|
||||
for path in sorted(seed_dir.glob("*.json")):
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
@@ -267,25 +294,69 @@ def load_seed_catalogs(seed_dir: Path = SEED_DIR,
|
||||
logger.error("Seed directory not found: %s", seed_dir)
|
||||
return {}
|
||||
|
||||
files = sorted(seed_dir.glob("*.json"))
|
||||
files = seed_catalog_paths(seed_dir)
|
||||
if only:
|
||||
wanted = [w.lower() for w in only]
|
||||
files = [f for f in files if any(w in f.name.lower() for w in wanted)]
|
||||
|
||||
# `only` is an explicit request for named files, so it wins over the
|
||||
# ACTIVE_BRANDS filter - that is how an archived brand can still be
|
||||
# re-ingested deliberately without first editing the config.
|
||||
respect_active = not only
|
||||
|
||||
brand_products: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
|
||||
skipped_inactive = 0
|
||||
for path in files:
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
logger.warning("Skipping %s - no brand/products found", path.name)
|
||||
continue
|
||||
resolved = resolve_parent_brand(data["brand"])
|
||||
if respect_active and not is_active_brand(resolved):
|
||||
skipped_inactive += 1
|
||||
continue
|
||||
brand_products[resolved].extend(data["products"])
|
||||
logger.info("Read %d products from %s -> resolved brand '%s'",
|
||||
len(data["products"]), path.name, resolved)
|
||||
|
||||
if skipped_inactive:
|
||||
logger.info("Skipped %d seed catalog(s) outside ACTIVE_BRANDS", skipped_inactive)
|
||||
|
||||
return dict(brand_products)
|
||||
|
||||
|
||||
def load_brand_products(brand: str) -> List[Dict[str, Any]]:
|
||||
"""Every seed product belonging to `brand`, resolved properly.
|
||||
|
||||
Use this instead of `load_seed_catalogs(only=[brand])` when you have a
|
||||
BRAND NAME. `only=` is a case-insensitive substring match on the FILE NAME,
|
||||
which is the right thing for `seed_sample_data.py --only amul` but the
|
||||
wrong thing for a brand:
|
||||
|
||||
* "Hindustan Unilever" contains a space; the file is
|
||||
`brand_catalog_hindustan_unilever.json`. The substring never matches
|
||||
and you silently get zero products - no error, just an empty result.
|
||||
* Even with the slug, a filename match misses the other files that feed
|
||||
the same table: `brand_catalog_tata.json` holds 121 Hindustan Unilever
|
||||
products because the "hul tata tea" alias claims it.
|
||||
|
||||
Going through `index_seed_files()` fixes both: it is keyed by the resolved
|
||||
table suffix and its value is the full list of files feeding that table.
|
||||
Archived catalogs are included, so an explicitly requested brand is found
|
||||
whether or not it is currently active.
|
||||
"""
|
||||
index = index_seed_files()
|
||||
products: List[Dict[str, Any]] = []
|
||||
for path in index.get(brand_slug(brand), []):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
products.extend(data["products"])
|
||||
logger.info("Read %d products from %s for brand '%s'",
|
||||
len(data["products"]), path.name, brand)
|
||||
return products
|
||||
|
||||
|
||||
def seed_brands(brand_products: Dict[str, List[Dict[str, Any]]], cleanup: bool = True) -> int:
|
||||
"""Upsert grouped products into their brand tables. Returns the row count."""
|
||||
total = 0
|
||||
@@ -339,7 +410,7 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
|
||||
by_slug: Dict[str, set] = defaultdict(set)
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
for path in sorted(seed_dir.glob("*.json")):
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
@@ -354,6 +425,9 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
|
||||
def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
|
||||
"""Repair the table <-> seed-file correspondence in both directions.
|
||||
|
||||
Scoped to ACTIVE_BRANDS when that is set: an archived brand is neither
|
||||
exported nor seeded, in either direction.
|
||||
|
||||
Idempotent and non-destructive:
|
||||
|
||||
* A populated table with no seed file gets one exported.
|
||||
@@ -373,6 +447,22 @@ def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
|
||||
file_index = index_seed_files()
|
||||
collisions = _detect_collisions()
|
||||
|
||||
# BOTH SIDES MUST BE NARROWED TO THE ACTIVE BRANDS, OR NEITHER.
|
||||
#
|
||||
# _db_brand_counts() goes through _list_brand_table_suffixes(), so under
|
||||
# ACTIVE_BRANDS it only sees the active tables. index_seed_files() reads the
|
||||
# archive directory too, deliberately, so that re-activating a brand needs
|
||||
# only a config change.
|
||||
#
|
||||
# Left mismatched, every archived brand looks like "a seed file whose table
|
||||
# is empty" and lands in `to_seed` - so the boot reconcile would re-seed all
|
||||
# 27 archived catalogs on the next restart, recreate their tables, and
|
||||
# silently undo the archiving. Verified: a dry run reported exactly that.
|
||||
if active_brand_suffixes() is not None:
|
||||
file_index = {
|
||||
slug: paths for slug, paths in file_index.items() if is_active_suffix(slug)
|
||||
}
|
||||
|
||||
to_export = sorted(
|
||||
suffix for suffix, count in db_counts.items()
|
||||
if count > 0 and suffix not in file_index
|
||||
|
||||
@@ -24,8 +24,25 @@ from app.services.discount_service import predict_discounts_for_store
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Every model this module knows how to train. Still the full set accepted by
|
||||
# scripts/train_ml_models.py and POST /api/admin/store-intelligence/train.
|
||||
ALL_MODELS = ["discount", "trending", "popularity", "forecast", "store_performance", "purchase_propensity"]
|
||||
|
||||
# What train_all() trains when the caller does not name a subset.
|
||||
#
|
||||
# `forecast`, `store_performance` and `purchase_propensity` are omitted
|
||||
# deliberately: all three fit cleanly, but nothing reads their output back.
|
||||
# store_db.get_latest_demand_forecast() has zero callers, and neither of the
|
||||
# other two has an inference consumer anywhere in the app or the MCP tools.
|
||||
# Training them by default spent roughly half of every retrain, and ~5MB of
|
||||
# artifacts, producing bundles no request would ever load.
|
||||
#
|
||||
# Their training code, CLI flags and API options are untouched - ask for one by
|
||||
# name (`--models store_performance`) and it trains and is written to
|
||||
# MODEL_ARTIFACTS_DIR exactly as before. Move a model back into this list once
|
||||
# something actually serves it.
|
||||
PRODUCTION_MODELS = ["discount", "trending", "popularity"]
|
||||
|
||||
|
||||
def _load_common_frames():
|
||||
order_items = store_db.get_order_items_df()
|
||||
@@ -144,7 +161,11 @@ def train_purchase_propensity(orders: pd.DataFrame) -> Dict:
|
||||
|
||||
|
||||
def train_all(models: Optional[List[str]] = None) -> Dict[str, Dict]:
|
||||
models = models or ALL_MODELS
|
||||
"""Train `models`, defaulting to the models that are actually served.
|
||||
|
||||
Pass an explicit list (including any of ALL_MODELS) to train more.
|
||||
"""
|
||||
models = models or PRODUCTION_MODELS
|
||||
order_items, orders, store_products = _load_common_frames()
|
||||
results: Dict[str, Dict] = {}
|
||||
|
||||
|
||||
@@ -205,8 +205,16 @@ def _brand_index() -> tuple[list[str], Dict[str, str]]:
|
||||
if _MENTION_CACHE["map"] is not None and now - _MENTION_CACHE["at"] < BRAND_MENTION_TTL_SECONDS:
|
||||
return _MENTION_CACHE["known"], _MENTION_CACHE["map"]
|
||||
|
||||
known = list(KNOWN_BRANDS)
|
||||
mapping = dict(BRAND_SEARCH_MAP)
|
||||
# KNOWN_BRANDS/BRAND_SEARCH_MAP are a hand-tuned static table of 16 brands.
|
||||
# Under ACTIVE_BRANDS they must be narrowed too, not just added to: without
|
||||
# this a query for an archived brand ("Nestle") still parses as a brand
|
||||
# mention, gets routed to brand_catalog mode, and returns zero rows - the
|
||||
# same silent wrong-result failure mode as fuzzy category detection. Dropped
|
||||
# here, it is treated as ordinary search text instead.
|
||||
from app.services.active_brands import is_active_brand
|
||||
|
||||
known = [b for b in KNOWN_BRANDS if is_active_brand(b)]
|
||||
mapping = {k: v for k, v in BRAND_SEARCH_MAP.items() if is_active_brand(v)}
|
||||
try:
|
||||
from app.services.vector_store import list_available_brands
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ from app.infrastructure.settings import (
|
||||
DB_CONNECT_TIMEOUT_SECONDS,
|
||||
)
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.active_brands import filter_suffixes, is_active_suffix
|
||||
|
||||
|
||||
from app.services.s3_service import s3_service
|
||||
@@ -562,6 +563,12 @@ def list_available_brands() -> List[str]:
|
||||
# Extract brand name from table name (e.g., brand_britannia -> britannia)
|
||||
table_name = table[0]
|
||||
suffix = table_name[len("brand_"):].lower() if table_name.startswith("brand_") else table_name.lower()
|
||||
# ACTIVE_BRANDS narrows the whole app here: this list feeds
|
||||
# /api/brands, the MCP list_brands tool, rag_service,
|
||||
# query_intent's brand index, nutrition enrichment, store_db
|
||||
# and get_products_all_brands. The table itself is untouched.
|
||||
if not is_active_suffix(suffix):
|
||||
continue
|
||||
# Prefer the original display name from the brand map if available
|
||||
brands.append(display_name_for_suffix(suffix, brand_map))
|
||||
except Exception as e:
|
||||
@@ -1220,12 +1227,23 @@ def _table_exists(cur, table_name: str) -> bool:
|
||||
return bool(row and row[0])
|
||||
|
||||
|
||||
def _list_brand_table_suffixes(cur) -> List[str]:
|
||||
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table."""
|
||||
def _list_brand_table_suffixes(cur, *, include_inactive: bool = False) -> List[str]:
|
||||
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table.
|
||||
|
||||
Narrowed to ACTIVE_BRANDS by default. This is the shared chokepoint for
|
||||
get_brand_overview, semantic_search, text_search, lexical_search,
|
||||
list_all_categories and brand_sync._db_brand_counts, which is why the
|
||||
active-brand setting reaches all of them without touching any of them.
|
||||
|
||||
`include_inactive=True` is the escape hatch for callers that address an
|
||||
archived brand deliberately - an explicit re-ingest or a backfill - so
|
||||
narrowing the catalog never means losing the ability to maintain it.
|
||||
"""
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT table_name FROM information_schema.tables
|
||||
WHERE table_schema = 'public' AND table_name LIKE 'brand_%'
|
||||
"""
|
||||
)
|
||||
return [row[0][len("brand_"):] for row in cur.fetchall()]
|
||||
suffixes = [row[0][len("brand_"):] for row in cur.fetchall()]
|
||||
return suffixes if include_inactive else filter_suffixes(suffixes)
|
||||
|
||||
Reference in New Issue
Block a user