Add Dagster orchestration and reduce active brands in backend

This commit is contained in:
sriram
2026-08-20 16:39:54 +05:30
parent fbb1356e47
commit 7bf8dc6922
66 changed files with 2664 additions and 21 deletions

View File

@@ -56,6 +56,17 @@ os.environ.setdefault("DB_PASSWORD", "test-password-not-real")
os.environ.setdefault("USE_S3", "false")
os.environ.setdefault("USE_GOOGLE_CSE", "false")
# Unconditional, NOT setdefault. The suite pins brand-name behaviour all over
# the place ("any Nestle chocolates?", the suggest ranking fixtures), and a
# developer's backend/.env now carries a real ACTIVE_BRANDS value. Letting that
# leak in would make those tests pass or fail depending on whose machine ran
# them - the same trap AUTH_ALLOW_ANY_LOGIN sprang before it was pinned here.
#
# Blank means "no brand filtering", so every existing test sees the historical
# behaviour. The filtering itself is covered by tests/test_active_brands.py,
# which sets the value explicitly and clears the parsed cache.
os.environ["ACTIVE_BRANDS"] = ""
# Auth is set unconditionally (not setdefault): the suite asserts on the real
# guards, so it must never inherit a developer's AUTH_ENABLED=false.
os.environ["AUTH_ENABLED"] = "true"

247
tests/test_active_brands.py Normal file
View File

@@ -0,0 +1,247 @@
"""ACTIVE_BRANDS: the one setting that narrows the whole application.
conftest.py pins ACTIVE_BRANDS="" for the rest of the suite, so this module
sets it explicitly and restores it. Everything here is pure config parsing -
no database, no network.
"""
from __future__ import annotations
import pytest
from app.services import active_brands as ab
from app.services.brand_sync import ARCHIVE_DIR_NAME, SEED_DIR, seed_catalog_paths
@pytest.fixture
def configured(monkeypatch):
"""Set ACTIVE_BRANDS and drop the parsed cache on the way in and out."""
def _apply(raw: str):
monkeypatch.setattr(ab, "_RAW_ACTIVE_BRANDS", raw)
ab.invalidate()
return ab
yield _apply
ab.invalidate()
THREE = "Amul,Cadbury,Hindustan Unilever"
# --- the empty-means-all contract -------------------------------------------
# Production leaves ACTIVE_BRANDS unset. If blank ever started meaning "no
# brands are active" instead of "every brand is active", the entire catalog
# would go empty in production and the API would keep answering 200s while
# doing it - the single most expensive failure mode this project has had.
@pytest.mark.parametrize("blank", ["", " ", ",", " , , "])
def test_blank_config_disables_filtering_entirely(configured, blank):
cfg = configured(blank)
assert cfg.filtering_enabled() is False
assert cfg.active_brand_suffixes() is None
assert cfg.is_active_suffix("nestle") is True
assert cfg.is_active_brand("literally anything") is True
assert cfg.filter_suffixes(["amul", "nestle", "p_g"]) == ["amul", "nestle", "p_g"]
def test_configured_brands_narrow_the_set(configured):
cfg = configured(THREE)
assert cfg.filtering_enabled() is True
assert cfg.active_brand_suffixes() == frozenset({"amul", "cadbury", "hindustan_unilever"})
assert cfg.active_display_names() == ["Amul", "Cadbury", "Hindustan Unilever"]
def test_filter_suffixes_drops_inactive_and_keeps_order(configured):
cfg = configured(THREE)
given = ["nestle", "amul", "p_g", "hindustan_unilever", "cadbury", "pepsico"]
assert cfg.filter_suffixes(given) == ["amul", "hindustan_unilever", "cadbury"]
# --- names are resolved, not string-matched ---------------------------------
def test_names_resolve_through_the_brand_aliases(configured):
"""Config must agree with the storage layer about which table a brand is.
"Tata" has no table of its own - the "hul tata tea" alias routes it into
brand_hindustan_unilever, where brand_catalog_tata.json's 121 products
already live. Comparing raw strings here would have let ACTIVE_BRANDS and
resolve_parent_brand disagree, hiding those products from an active brand.
"""
cfg = configured(THREE)
assert cfg.is_active_brand("Tata") is True
assert cfg.is_active_brand("tata tea") is True
assert cfg.is_active_brand("Dove") is True # -> hindustan unilever
assert cfg.is_active_brand("cadbury dairy milk") is True
assert cfg.is_active_brand("amul butter") is True
assert cfg.is_active_brand("Nestle") is False
assert cfg.is_active_brand("P&G") is False
def test_configuring_a_sub_brand_activates_its_parent_table(configured):
cfg = configured("Tata")
assert cfg.active_brand_suffixes() == frozenset({"hindustan_unilever"})
def test_whitespace_and_duplicates_are_tolerated(configured):
cfg = configured(" Amul , amul , Cadbury ")
assert cfg.active_brand_suffixes() == frozenset({"amul", "cadbury"})
# --- the query_intent brand index must narrow too ---------------------------
def test_brand_index_drops_inactive_static_brands(configured, monkeypatch):
"""An archived brand must stop parsing as a brand mention.
KNOWN_BRANDS/BRAND_SEARCH_MAP are a hand-tuned static table, and the live
brand list only ever *added* to it. Left alone, "any Nestle chocolates?"
would still resolve to a brand, get routed to brand-catalog mode, and come
back empty rather than falling through to ordinary search.
"""
from app.services import query_intent as qi
configured(THREE)
monkeypatch.setattr(qi, "list_available_brands", lambda: [], raising=False)
qi.invalidate_brand_mention_cache()
monkeypatch.setattr(
"app.services.vector_store.list_available_brands", lambda: [], raising=False
)
known, mapping = qi._brand_index()
assert "Amul" in known
assert "Cadbury" in known
assert "Hindustan Unilever" in known
assert "Nestle" not in known
assert "Pepsico" not in known
# Synonyms follow their target brand.
assert mapping.get("hul") == "Hindustan Unilever"
assert "pepsi" not in mapping
assert "coke" not in mapping
qi.invalidate_brand_mention_cache()
def test_extract_brand_mention_ignores_an_inactive_brand(configured, monkeypatch):
from app.services import query_intent as qi
configured(THREE)
monkeypatch.setattr(
"app.services.vector_store.list_available_brands", lambda: [], raising=False
)
qi.invalidate_brand_mention_cache()
assert qi.extract_brand_mention("show me Amul butter") == "Amul"
assert qi.extract_brand_mention("any Nestle chocolates?") is None
qi.invalidate_brand_mention_cache()
# --- the seed archive -------------------------------------------------------
def test_archived_catalogs_are_still_discoverable_on_disk():
"""Re-activating a brand must be a config change, not a file move.
seed_catalog_paths() reads the archive subdirectory as well, so flipping a
name into ACTIVE_BRANDS is enough to seed and serve it again.
"""
names = {p.name for p in seed_catalog_paths(SEED_DIR)}
assert "brand_catalog_amul.json" in names
assert "brand_catalog_nestle.json" in names, "archived catalogs must stay reachable"
active_only = {p.name for p in SEED_DIR.glob("*.json")}
assert "brand_catalog_nestle.json" not in active_only, (
"an archived catalog must NOT be picked up by the plain glob the boot "
"auto-seed used to run over every file"
)
assert (SEED_DIR / ARCHIVE_DIR_NAME).is_dir()
def test_load_seed_catalogs_only_returns_active_brands(configured):
from app.services import brand_sync
configured(THREE)
grouped = brand_sync.load_seed_catalogs()
assert set(grouped) == {"amul", "cadbury", "hindustan unilever"}
# tata.json resolves into hindustan unilever, so its products must be there.
assert len(grouped["hindustan unilever"]) > 104
def test_explicit_only_overrides_the_active_filter(configured):
"""`only=` is a deliberate request, so it must still reach an archived brand."""
from app.services import brand_sync
configured(THREE)
grouped = brand_sync.load_seed_catalogs(only=["nestle"])
# resolve_parent_brand returns the lowercase alias parent, not the file's
# "Nestle" display casing.
assert set(grouped) == {"nestle"}
assert len(grouped["nestle"]) == 123
def test_reconcile_does_not_resurrect_archived_brands(configured, monkeypatch):
"""The boot reconcile must not re-seed the brands we just archived.
REGRESSION. `_db_brand_counts()` is narrowed by ACTIVE_BRANDS (it goes
through _list_brand_table_suffixes), while `index_seed_files()` reads the
archive directory on purpose. Left mismatched, every archived catalog looks
like "a seed file whose table is empty" and lands in `to_seed` - so the
reconcile that runs on every API start, and again every 300 seconds, would
recreate all 27 archived brand tables and undo the archiving silently.
A dry run confirmed exactly that before the fix.
"""
from app.services import brand_sync
configured(THREE)
# Only the active tables are visible, which is what the real filtered
# _db_brand_counts() returns.
monkeypatch.setattr(
brand_sync,
"_db_brand_counts",
lambda: {"amul": 122, "cadbury": 105, "hindustan_unilever": 225},
)
summary = brand_sync.reconcile_brand_catalogs(dry_run=True)
seeded = summary.get("would_seed") or summary.get("seeded") or []
exported = summary.get("would_export") or summary.get("exported") or []
assert seeded == [], "archived brands must never be re-seeded by reconcile"
assert exported == []
assert summary["files"] == 3, "reconcile must only consider active seed files"
@pytest.mark.parametrize(
"brand,expected_min",
[("Amul", 122), ("Cadbury", 105), ("Hindustan Unilever", 225), ("Tata", 225)],
)
def test_load_brand_products_resolves_multi_word_brands(brand, expected_min):
"""REGRESSION: a brand name is not a filename.
`load_seed_catalogs(only=["Hindustan Unilever"])` matches substrings of the
FILE NAME, and "hindustan unilever" (space) is not a substring of
"brand_catalog_hindustan_unilever.json" (underscore). It returned zero
products, silently - the Dagster partition for HUL reported RUN_SUCCESS
having ingested nothing.
`load_brand_products` goes through the seed-file index instead, which is
keyed by resolved table suffix, so it also picks up brand_catalog_tata.json
(121 HUL products via the "hul tata tea" alias).
"""
from app.services.brand_sync import load_brand_products
assert len(load_brand_products(brand)) >= expected_min
def test_load_brand_products_reaches_archived_catalogs():
"""An explicitly named brand must be found even while it is archived."""
from app.services.brand_sync import load_brand_products
assert len(load_brand_products("Nestle")) == 123
def test_load_seed_catalogs_only_still_matches_filenames():
"""The `only=` filename semantics are unchanged - the CLI documents them."""
from app.services.brand_sync import load_seed_catalogs
grouped = load_seed_catalogs(only=["amul"])
assert sum(len(v) for v in grouped.values()) == 122

View File

@@ -17,15 +17,22 @@ from pathlib import Path
import pytest
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.brand_sync import seed_catalog_paths
from app.services.vector_store import _sanitize_name
SEED_DIR = Path(__file__).resolve().parents[1] / "data" / "seed_catalogs"
def _seed_brand_fields() -> list[str]:
"""The `brand` value of every seed catalog (skipping non-catalog files)."""
"""The `brand` value of every seed catalog (skipping non-catalog files).
Deliberately covers the `archive/` subdirectory as well as the active
catalogs. Archiving a brand takes it out of the running app, but it must
not take it out of this guarantee - an archived file is re-activated by a
config change alone, and it has to land in the same table it always did.
"""
brands = []
for path in sorted(SEED_DIR.glob("*.json")):
for path in seed_catalog_paths(SEED_DIR):
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except Exception:
@@ -36,29 +43,45 @@ def _seed_brand_fields() -> list[str]:
return brands
# Every seed catalog and the table it must continue to feed. `tata` mapping to
# hindustan_unilever is not a typo: alias "hul tata tea" claims it, which is
# where all 121 of that file's products already live.
# Every seed catalog and the table it must continue to feed, active and
# archived alike. `tata` mapping to hindustan_unilever is not a typo: alias
# "hul tata tea" claims it, which is where all 121 of that file's products
# already live - and it is why brand_catalog_tata.json stays in the active
# directory while Hindustan Unilever is an active brand.
EXPECTED_SEED_TABLES = {
"aachi": "brand_aachi",
"amul": "brand_amul",
"anil": "brand_anil",
"bikaji": "brand_bikaji",
"britannia": "brand_britannia",
"cadbury": "brand_cadbury",
"cavinkare": "brand_cavinkare",
"coca-cola": "brand_coca_cola",
"colgate-palmolive": "brand_colgate_palmolive",
"dabur": "brand_dabur",
"everest": "brand_everest",
"fortune": "brand_fortune",
"godrej": "brand_godrej",
"grb": "brand_grb",
"haldirams": "brand_haldirams",
"hindustan unilever": "brand_hindustan_unilever",
# Ingested straight into the database with no BRAND_ALIASES entry, so it
# exercises the unaliased path: resolve_parent_brand returns it unchanged
# and it gets its own table. Pinned here to catch the day some new alias
# whole-word-matches "idhayam" and silently re-parents 24 products.
"idhayam": "brand_idhayam",
"itc": "brand_itc",
"kaleesuwari": "brand_kaleesuwari",
"lion dates": "brand_lion_dates",
"Manna": "brand_manna",
"marico": "brand_marico",
"mdh": "brand_mdh",
"milky mist": "brand_milky_mist",
"mtr": "brand_mtr",
"naga": "brand_naga",
"Nestle": "brand_nestle",
"p&g": "brand_p_g",
"parle": "brand_parle",
"pepsico": "brand_pepsico",
"tata": "brand_hindustan_unilever",
}

View File

@@ -0,0 +1,219 @@
"""The Dagster definitions load, and the graph is the one we meant to build.
Skipped entirely when dagster is not installed: it lives in
requirements-orchestration.txt, not requirements.txt, so the API image and CI
runs that only install the app dependencies must still get a green suite.
Nothing here executes a run. These are structural assertions - that the code
location imports, that the lineage edges exist, that the schedules are off,
and that the write guard refuses a remote database. A broken code location is
the failure mode worth catching early, because the Dagster webserver reports
it as an opaque load error long after the change that caused it.
"""
import os
import pytest
dagster = pytest.importorskip("dagster", reason="orchestration extra not installed")
@pytest.fixture(scope="module")
def defs():
from orchestration.definitions import defs as _defs
return _defs
@pytest.fixture(scope="module")
def asset_graph(defs):
return defs.get_repository_def().asset_graph
def _keys(asset_graph):
return {key.to_user_string() for key in asset_graph.get_all_asset_keys()}
def test_code_location_loads(defs):
assert defs is not None
def test_every_expected_asset_exists(asset_graph):
assert _keys(asset_graph) == {
# catalog
"active_brand",
"raw_products",
"validated_products",
"enriched_products",
"catalog_database",
"product_embeddings",
"vector_index",
# nutrition
"nutrition_data",
"nutrition_models",
# ml
"training_dataset",
"trained_models",
"model_evaluation",
}
@pytest.mark.parametrize(
"asset_key,expected_parents",
[
("raw_products", {"active_brand"}),
("validated_products", {"active_brand", "raw_products"}),
("enriched_products", {"active_brand", "validated_products"}),
("catalog_database", {"active_brand", "enriched_products"}),
("product_embeddings", {"active_brand", "catalog_database"}),
("vector_index", {"active_brand", "product_embeddings"}),
("nutrition_data", {"catalog_database"}),
("nutrition_models", {"nutrition_data"}),
("trained_models", {"training_dataset"}),
("model_evaluation", {"trained_models"}),
],
)
def test_lineage_edges(asset_graph, asset_key, expected_parents):
"""The DAG shape IS the deliverable - pin it.
Losing an edge does not fail a run, it just silently lets an asset
materialize against stale upstream data (embedding rows that were never
written, models fitted on an empty orders table).
"""
from dagster import AssetKey
node = asset_graph.get(AssetKey(asset_key))
assert {k.to_user_string() for k in node.parent_keys} == expected_parents
def test_catalog_assets_are_partitioned_by_brand(asset_graph):
"""Per-brand partitioning is what bounds memory and lets one brand retry."""
from dagster import AssetKey
for key in (
"active_brand",
"raw_products",
"validated_products",
"enriched_products",
"catalog_database",
"product_embeddings",
"vector_index",
):
assert asset_graph.get(AssetKey(key)).is_partitioned, key
def test_cross_brand_assets_are_not_partitioned(asset_graph):
"""nutrition_models and the ML models fit across every brand at once.
Partitioning them would produce per-brand indexes that answer a narrower
question than "find a healthier alternative" actually asks.
"""
from dagster import AssetKey
for key in ("nutrition_models", "training_dataset", "trained_models", "model_evaluation"):
assert not asset_graph.get(AssetKey(key)).is_partitioned, key
def test_all_four_jobs_resolve(defs):
assert {job.name for job in defs.jobs} == {
"catalog_ingestion_job",
"embedding_refresh_job",
"nutrition_enrichment_job",
"ml_training_job",
}
def test_every_schedule_ships_stopped(defs):
"""A schedule that auto-starts would run heavy jobs on an 8GB dev laptop.
This is the assertion that keeps `dagster dev` from quietly becoming a
background workload.
"""
from dagster import DefaultScheduleStatus
assert defs.schedules, "expected schedules to be defined"
for schedule in defs.schedules:
assert schedule.default_status == DefaultScheduleStatus.STOPPED, schedule.name
def test_sensor_ships_stopped_and_is_not_hot(defs):
from dagster import DefaultSensorStatus
assert defs.sensors, "expected the seed-catalog sensor"
for sensor in defs.sensors:
assert sensor.default_status == DefaultSensorStatus.STOPPED, sensor.name
assert sensor.minimum_interval_seconds >= 60, sensor.name
def test_network_assets_retry_and_are_bounded(asset_graph):
"""Retries must exist on the flaky steps and must never be unbounded."""
from dagster import AssetKey
for key in ("raw_products", "enriched_products", "product_embeddings"):
policy = asset_graph.get(AssetKey(key)).assets_def.op.retry_policy
assert policy is not None, key
assert 0 < policy.max_retries <= 3, key
# --- the write guard --------------------------------------------------------
def test_guard_allows_a_local_database(monkeypatch):
from app.infrastructure import settings
from orchestration import config
monkeypatch.setattr(settings, "DB_HOST", "localhost")
monkeypatch.setattr(settings, "DB_PORT", "5432")
assert config.require_local_database("test") == "localhost:5432"
def test_guard_refuses_a_remote_database(monkeypatch):
"""backend/.env points at production. This is the last line of defence.
If the env-file ordering in definitions.py is ever broken, this turns a
silent write to the live catalog into a red run with an explanation.
"""
from dagster import Failure
from app.infrastructure import settings
from orchestration import config
monkeypatch.setattr(settings, "DB_HOST", "31.97.228.132")
monkeypatch.setattr(settings, "DB_PORT", "6054")
monkeypatch.delenv("ORCHESTRATION_ALLOW_REMOTE_WRITES", raising=False)
with pytest.raises(Failure) as excinfo:
config.require_local_database("catalog_database")
assert "31.97.228.132" in str(excinfo.value)
def test_guard_can_be_overridden_deliberately(monkeypatch):
from app.infrastructure import settings
from orchestration import config
monkeypatch.setattr(settings, "DB_HOST", "31.97.228.132")
monkeypatch.setattr(settings, "DB_PORT", "6054")
monkeypatch.setenv("ORCHESTRATION_ALLOW_REMOTE_WRITES", "true")
assert config.require_local_database("catalog_database") == "31.97.228.132:6054"
# --- brand config passthrough ----------------------------------------------
def test_partitions_follow_active_brands(monkeypatch):
"""3 brands -> 3 partitions. Widening the working set needs no code edit."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "_RAW_ACTIVE_BRANDS", "Amul,Cadbury,Nestle")
active_brands.invalidate()
try:
from orchestration.partitions import active_brand_names
assert active_brand_names() == ["Amul", "Cadbury", "Nestle"]
finally:
active_brands.invalidate()
def test_resolve_brands_prefers_explicit_run_config(monkeypatch):
from orchestration.config import resolve_brands
assert resolve_brands(["Britannia"]) == ["Britannia"]