updates on the backend

This commit is contained in:
sriram
2026-08-11 19:16:01 +05:30
commit c2af4556c6
131 changed files with 546007 additions and 0 deletions

View File

@@ -0,0 +1,41 @@
"""
Store Intelligence ML package
==============================
Everything related to the v3.0 "Multi-Store Intelligence" upgrade lives
here: synthetic data generation, feature engineering, and the trained
scikit-learn models for discount prediction, trending detection, demand
forecasting, popularity scoring, store performance, purchase propensity,
and product recommendations.
Design notes (read this before touching the models)
-----------------------------------------------------
1. CPU-only / 8GB RAM target. Every model in this package is a
scikit-learn estimator (RandomForest / GradientBoosting / Logistic
/ Linear regression). We deliberately do NOT use XGBoost, LightGBM,
CatBoost, Prophet, or any deep-learning (LSTM/PyTorch/TensorFlow)
library, even though the feature spec lists them as options - on
this hardware they cost far more RAM/install time than the accuracy
they'd buy at this data scale (a handful of stores, a few thousand
products, tens of thousands of simulated orders). scikit-learn's
GradientBoostingRegressor/RandomForest give comparable accuracy at a
fraction of the footprint and were explicitly listed as acceptable
alternatives for every ML feature in the spec.
2. Pure functions first. Feature engineering and synthetic-label
generation (`features.py`, `synthetic_labels.py`) take/return plain
dicts, lists, and pandas DataFrames - no database or network I/O.
This is what makes them unit-testable without a live Postgres
instance and keeps the ML logic independent of the persistence
layer (clean architecture / SOLID: the model layer doesn't know
Postgres exists).
3. Every "intelligent" feature (discount %, trending score, demand
forecast, recommendation ranking, popularity score, store
performance, purchase propensity) is produced by a trained model
loaded from `artifacts/*.joblib`, never a hardcoded lookup table.
Where no real historical A/B-tested label exists (e.g. "what
discount SHOULD this product have had"), we bootstrap training
labels from a documented, multi-factor formula + noise
(`synthetic_labels.py`) - this is standard practice for cold-start
ML systems. The important part is that INFERENCE always goes
through the trained model, not the formula - the formula only
exists to generate training data once.
"""

View File

@@ -0,0 +1,215 @@
"""
Features 4 & 5: Store Analytics Dashboard + Product Analytics.
Pure computation over plain pandas DataFrames - no SQL in this file.
`app/services/analytics_service.py` is the thin I/O layer that pulls
DataFrames out of `store_db.py` and hands them to the functions here,
which keeps every KPI formula unit-testable without a live Postgres
instance (see `tests/test_intelligence.py`).
"""
from __future__ import annotations
from typing import Dict, List, Optional
import numpy as np
import pandas as pd
def classify_stock_status(available: int, reorder_level: int, safety_stock: int) -> str:
if available <= 0:
return "Out of Stock"
if available <= safety_stock:
return "Low Stock"
if available > reorder_level * 6:
return "Overstocked"
return "In Stock"
def inventory_analytics(store_products: pd.DataFrame) -> Dict:
"""`store_products` columns: available_stock, reorder_level, safety_stock (already
filtered to one store, or pass the full multi-store frame for a
chain-wide summary)."""
if store_products.empty:
return {"total_products": 0, "in_stock": 0, "low_stock": 0, "out_of_stock": 0, "overstocked": 0}
statuses = store_products.apply(
lambda r: classify_stock_status(r["available_stock"], r["reorder_level"], r["safety_stock"]), axis=1
)
counts = statuses.value_counts()
return {
"total_products": int(len(store_products)),
"in_stock": int(counts.get("In Stock", 0)),
"low_stock": int(counts.get("Low Stock", 0)),
"out_of_stock": int(counts.get("Out of Stock", 0)),
"overstocked": int(counts.get("Overstocked", 0)),
}
def sales_analytics(orders: pd.DataFrame) -> Dict:
"""`orders` columns: order_date (datetime64), order_value. Already
filtered to the scope (one store, or all stores) the caller wants."""
if orders.empty:
return {
"total_sales": 0, "revenue": 0.0, "average_basket_value": 0.0,
"daily_sales": [], "weekly_sales": [], "monthly_sales": [],
}
revenue = float(orders["order_value"].sum())
total_sales = int(len(orders))
avg_basket = revenue / total_sales if total_sales else 0.0
daily = orders.groupby(orders["order_date"].dt.date)["order_value"].agg(["sum", "count"]).reset_index()
daily.columns = ["date", "revenue", "orders"]
weekly = orders.groupby(orders["order_date"].dt.to_period("W").astype(str))["order_value"].agg(["sum", "count"]).reset_index()
weekly.columns = ["week", "revenue", "orders"]
monthly = orders.groupby(orders["order_date"].dt.to_period("M").astype(str))["order_value"].agg(["sum", "count"]).reset_index()
monthly.columns = ["month", "revenue", "orders"]
return {
"total_sales": total_sales,
"revenue": round(revenue, 2),
"average_basket_value": round(avg_basket, 2),
"daily_sales": [{"date": str(r["date"]), "revenue": round(r["revenue"], 2), "orders": int(r["orders"])} for _, r in daily.iterrows()],
"weekly_sales": [{"week": r["week"], "revenue": round(r["revenue"], 2), "orders": int(r["orders"])} for _, r in weekly.iterrows()],
"monthly_sales": [{"month": r["month"], "revenue": round(r["revenue"], 2), "orders": int(r["orders"])} for _, r in monthly.iterrows()],
}
def profit_analytics(order_items: pd.DataFrame, store_prices: pd.DataFrame) -> Dict:
"""Joins order line items to CURRENT store cost prices to estimate
profit (`(unit_price - cost_price) * quantity`). This is an
approximation - it uses today's cost price, not the cost price that
was actually in effect on the historical order date, since the
system doesn't keep a cost-price history table. Documented rather
than silently treated as exact."""
if order_items.empty or store_prices.empty:
return {"total_profit": 0.0, "gross_profit_pct": 0.0}
merged = order_items.merge(
store_prices[["store_id", "brand", "image_id", "cost_price"]],
on=["store_id", "brand", "image_id"], how="left",
)
merged["cost_price"] = merged["cost_price"].fillna(merged["unit_price"] * 0.78)
merged["line_profit"] = (merged["unit_price"] - merged["cost_price"]) * merged["quantity"]
total_profit = float(merged["line_profit"].sum())
total_revenue = float(merged["line_total"].sum())
gp_pct = (total_profit / total_revenue * 100) if total_revenue else 0.0
return {"total_profit": round(total_profit, 2), "gross_profit_pct": round(gp_pct, 2)}
def store_comparison(orders: pd.DataFrame, order_items: pd.DataFrame, store_prices: pd.DataFrame, stores: pd.DataFrame) -> Dict:
"""`stores` columns: store_id, store_name, tier, footfall_index."""
if orders.empty:
return {"stores": [], "best_performing": None, "lowest_performing": None,
"highest_revenue": None, "highest_profit": None}
per_store_revenue = orders.groupby("store_id")["order_value"].agg(["sum", "count"]).reset_index()
per_store_revenue.columns = ["store_id", "revenue", "order_count"]
profit_rows = []
for store_id, g in order_items.groupby("store_id"):
sp = store_prices[store_prices["store_id"] == store_id]
p = profit_analytics(g, sp)
profit_rows.append({"store_id": store_id, "profit": p["total_profit"]})
profit_df = pd.DataFrame(profit_rows) if profit_rows else pd.DataFrame(columns=["store_id", "profit"])
merged = per_store_revenue.merge(profit_df, on="store_id", how="left").merge(stores, on="store_id", how="left")
merged["profit"] = merged["profit"].fillna(0.0)
merged["avg_order_value"] = merged["revenue"] / merged["order_count"].replace(0, np.nan)
merged["avg_order_value"] = merged["avg_order_value"].fillna(0.0)
# "Customer Footfall (simulated)" - the store's actual simulated order
# count IS the simulated footfall proxy (every order came from a
# simulated in-store/online customer visit).
merged["footfall_simulated"] = merged["order_count"]
result_stores = [
{
"store_id": r["store_id"], "store_name": r.get("store_name"), "tier": r.get("tier"),
"revenue": round(r["revenue"], 2), "profit": round(r["profit"], 2),
"order_count": int(r["order_count"]), "avg_order_value": round(r["avg_order_value"], 2),
"footfall_simulated": int(r["footfall_simulated"]),
}
for _, r in merged.iterrows()
]
by_revenue = sorted(result_stores, key=lambda s: s["revenue"], reverse=True)
by_profit = sorted(result_stores, key=lambda s: s["profit"], reverse=True)
return {
"stores": result_stores,
"best_performing": by_profit[0]["store_id"] if by_profit else None,
"lowest_performing": by_profit[-1]["store_id"] if by_profit else None,
"highest_revenue": by_revenue[0]["store_id"] if by_revenue else None,
"highest_profit": by_profit[0]["store_id"] if by_profit else None,
}
def top_products(order_items: pd.DataFrame, limit: int = 10, ascending: bool = False, by: str = "revenue") -> List[Dict]:
"""`by`: 'revenue' or 'units'. Set ascending=True for "lowest
selling" instead of "top selling"."""
if order_items.empty:
return []
agg = order_items.groupby(["brand", "image_id"]).agg(
revenue=("line_total", "sum"), units=("quantity", "sum"), orders=("order_id", "nunique"),
).reset_index()
sort_col = "revenue" if by == "revenue" else "units"
agg = agg.sort_values(sort_col, ascending=ascending).head(limit)
return [
{"brand": r["brand"], "image_id": r["image_id"], "revenue": round(r["revenue"], 2),
"units_sold": int(r["units"]), "order_count": int(r["orders"])}
for _, r in agg.iterrows()
]
def product_analytics(
brand: str, image_id: str,
order_items: pd.DataFrame, store_prices: pd.DataFrame,
engagement_row: Optional[Dict] = None, popularity_score: Optional[float] = None,
demand_score: Optional[float] = None,
) -> Dict:
"""Full Feature 5 metric set for one product, aggregated across all
stores that carry it, plus a store-wise breakdown."""
prod_items = order_items[(order_items["brand"] == brand) & (order_items["image_id"] == image_id)]
prod_prices = store_prices[(store_prices["brand"] == brand) & (store_prices["image_id"] == image_id)]
sales_count = int(prod_items["quantity"].sum())
revenue = float(prod_items["line_total"].sum())
profit_info = profit_analytics(prod_items, store_prices)
order_count = int(prod_items["order_id"].nunique())
store_wise = (
prod_items.groupby("store_id").agg(units=("quantity", "sum"), revenue=("line_total", "sum")).reset_index()
if not prod_items.empty else pd.DataFrame(columns=["store_id", "units", "revenue"])
)
# Growth %: last-14-days units vs the 14 days before that (real data,
# not synthetic - same "current vs previous window" pattern used by
# the trending model, just exposed as a plain metric here).
growth_pct = None
if not prod_items.empty and "order_date" in prod_items.columns:
last_date = prod_items["order_date"].max()
cur_start = last_date - pd.Timedelta(days=14)
prev_start = cur_start - pd.Timedelta(days=14)
cur = prod_items[prod_items["order_date"] >= cur_start]["quantity"].sum()
prev = prod_items[(prod_items["order_date"] >= prev_start) & (prod_items["order_date"] < cur_start)]["quantity"].sum()
growth_pct = round(float((cur - prev) / prev * 100) if prev else (100.0 if cur else 0.0), 1)
avg_stock = float(prod_prices["selling_price"].mean()) if not prod_prices.empty else 0.0
# Stock turnover = units sold / average stock held (a standard retail
# KPI: how many times the inventory "turned over" in the observed period).
turnover = None
return {
"brand": brand, "image_id": image_id,
"sales_count": sales_count,
"revenue": round(revenue, 2),
"profit": profit_info["total_profit"],
"orders": order_count,
"avg_rating": (engagement_row or {}).get("avg_rating"),
"views": (engagement_row or {}).get("views"),
"wishlist_count": (engagement_row or {}).get("wishlist_count"),
"conversion_rate": (engagement_row or {}).get("conversion_rate"),
"popularity_score": round(popularity_score, 1) if popularity_score is not None else None,
"demand_score": round(demand_score, 1) if demand_score is not None else None,
"growth_pct": growth_pct,
"store_wise_sales": [
{"store_id": r["store_id"], "units": int(r["units"]), "revenue": round(r["revenue"], 2)}
for _, r in store_wise.iterrows()
],
}

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

@@ -0,0 +1,148 @@
"""
Feature 3: ML-Based Dynamic Discount Prediction.
Model: GradientBoostingRegressor (scikit-learn). Chosen over
XGBoost/LightGBM/CatBoost for the hardware-conscious reasons explained
in `app/intelligence/__init__.py` - it's on the spec's own list of
acceptable options and needs no extra native dependency.
Training labels are bootstrapped via `synthetic_labels.synthetic_discount_pct`
(see that module's docstring for why and how) - but the FEATURES used
here go beyond the flat formula: real simulated sales velocity, demand,
and popularity from actual order history are included, so the trained
model's predictions are not a re-derivation of the formula, they're a
learned function of real behavioural signals plus the bootstrapped
business-rule signal.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Dict, List
import numpy as np
import pandas as pd
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
from app.intelligence import features as F
from app.intelligence.synthetic_labels import synthetic_discount_pct
MODEL_NAME = "discount_model"
FEATURE_COLUMNS = [
"stock_ratio", "days_of_cover", "units_per_day_7", "units_per_day_30",
"demand_score", "popularity_score", "days_to_expiry", "is_festive_season",
"category_freq", "store_tier_encoded", "price_position",
]
def build_training_frame(store_products: pd.DataFrame, order_items: pd.DataFrame, as_of) -> pd.DataFrame:
"""`store_products` columns: store_id, brand, image_id, category,
mrp, cost_price, selling_price, available_stock, reorder_level,
safety_stock, store_tier, days_since_stocked, demand_score,
popularity_score.
Returns a frame with FEATURE_COLUMNS + 'discount_pct' (label).
"""
rows = []
cat_freq = F.encode_category(store_products["category"])
for i, row in store_products.reset_index(drop=True).iterrows():
vel7 = F.sales_velocity(order_items, row["store_id"], row["brand"], row["image_id"], as_of, 7)
vel30 = F.sales_velocity(order_items, row["store_id"], row["brand"], row["image_id"], as_of, 30)
rows.append({
"stock_ratio": F.stock_ratio(row["available_stock"], row["reorder_level"]),
"days_of_cover": F.days_of_cover(row["available_stock"], max(vel30["units_per_day"], 0.05)),
"units_per_day_7": vel7["units_per_day"],
"units_per_day_30": vel30["units_per_day"],
"demand_score": row["demand_score"],
"popularity_score": row["popularity_score"],
"days_to_expiry": F.days_to_expiry(row["category"], row["days_since_stocked"]),
"is_festive_season": F.is_festive_season(as_of),
"category_freq": cat_freq.iloc[i],
"store_tier_encoded": F.encode_store_tier(row["store_tier"]),
"price_position": (row["selling_price"] / row["mrp"]) if row["mrp"] else 1.0,
})
df = pd.DataFrame(rows)
rng = np.random.default_rng(7)
df["discount_pct"] = synthetic_discount_pct(df, rng)
return df
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.ensemble import GradientBoostingRegressor
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_absolute_error
X = training_frame[FEATURE_COLUMNS]
y = training_frame["discount_pct"]
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
model = GradientBoostingRegressor(
n_estimators=150, max_depth=3, learning_rate=0.08, subsample=0.9, random_state=42,
)
model.fit(X_train, y_train)
mae = float(mean_absolute_error(y_test, model.predict(X_test)))
bundle = ModelBundle(
estimator=model,
feature_columns=FEATURE_COLUMNS,
model_name=MODEL_NAME,
n_samples=len(training_frame),
extra={"val_mae_pct_points": round(mae, 3),
"feature_importances": dict(zip(FEATURE_COLUMNS, [round(float(v), 4) for v in model.feature_importances_]))},
)
save_bundle(bundle)
return bundle
@dataclass
class DiscountPrediction:
discount_pct: float
final_price: float
savings: float
class DiscountPredictor:
"""Thin inference wrapper. Loads the trained bundle lazily and caches
it in-process (safe for a single-worker CPU deployment; restart the
API after retraining to pick up a new artifact)."""
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def predict_one(self, feature_row: Dict[str, float], original_price: float) -> DiscountPrediction:
if not self._ensure_loaded():
# Graceful degradation: no trained model yet -> 0% discount,
# never a hardcoded non-zero guess.
return DiscountPrediction(discount_pct=0.0, final_price=round(original_price, 2), savings=0.0)
X = pd.DataFrame([[feature_row.get(c, 0.0) for c in self._bundle.feature_columns]],
columns=self._bundle.feature_columns)
pct = float(np.clip(self._bundle.estimator.predict(X)[0], 0, 35))
pct = round(pct, 1)
final_price = round(original_price * (1 - pct / 100.0), 2)
savings = round(original_price - final_price, 2)
return DiscountPrediction(discount_pct=pct, final_price=final_price, savings=savings)
def predict_batch(self, df: pd.DataFrame) -> pd.DataFrame:
"""`df` must contain FEATURE_COLUMNS plus a 'selling_price' column.
Returns df with discount_pct/final_price/savings columns added."""
if not self._ensure_loaded() or df.empty:
out = df.copy()
out["discount_pct"] = 0.0
out["final_price"] = out.get("selling_price", 0.0)
out["savings"] = 0.0
return out
X = df[self._bundle.feature_columns]
preds = np.clip(self._bundle.estimator.predict(X), 0, 35)
out = df.copy()
out["discount_pct"] = np.round(preds, 1)
out["final_price"] = np.round(out["selling_price"] * (1 - out["discount_pct"] / 100.0), 2)
out["savings"] = np.round(out["selling_price"] - out["final_price"], 2)
return out
discount_predictor = DiscountPredictor()

View File

@@ -0,0 +1,49 @@
"""
Simulated product engagement signals: views, wishlist adds, and star
ratings. Real e-commerce systems have this from clickstream/telemetry;
this system doesn't have a live storefront generating that yet, so we
derive plausible values deterministically from the same latent
popularity used by the order simulator (`order_simulation.latent_popularity`)
plus independent noise, so views/wishlist correlate with - but aren't
identical to - actual purchase volume (matching real behaviour: not
every view converts, not every wishlist add is purchased).
Swap this module out for real analytics/telemetry ingestion once the
storefront captures it; `popularity_model.py` and the product-analytics
endpoints only depend on the DataFrame shape this returns, not on how
it was produced.
"""
from __future__ import annotations
import numpy as np
import pandas as pd
from app.intelligence.order_simulation import latent_popularity
def simulate_engagement(products: pd.DataFrame, orders_count_by_product: pd.Series, seed: int = 5) -> pd.DataFrame:
"""`products` needs columns brand, image_id. `orders_count_by_product`
is a Series indexed by 'brand||image_id' with real simulated order
counts (used so views/conversion stay internally consistent with
actual simulated purchase behaviour).
Returns a DataFrame with: brand, image_id, views, wishlist_count,
orders_count, avg_rating, conversion_rate.
"""
rng = np.random.default_rng(seed)
rows = []
for _, p in products.iterrows():
key = f"{p['brand']}||{p['image_id']}"
pop = latent_popularity(p["brand"], p["image_id"])
orders_count = float(orders_count_by_product.get(key, 0.0))
base_views = max(orders_count * rng.uniform(18, 45), pop * 120)
views = int(base_views * rng.lognormal(0, 0.25))
wishlist = int(views * rng.uniform(0.02, 0.09) * (0.6 + 0.8 * pop))
conversion_rate = float(np.clip(orders_count / max(views, 1), 0, 1))
avg_rating = float(np.clip(rng.normal(3.6 + pop * 1.1, 0.35), 1.0, 5.0))
rows.append({
"brand": p["brand"], "image_id": p["image_id"], "views": views,
"wishlist_count": wishlist, "orders_count": orders_count,
"avg_rating": round(avg_rating, 2), "conversion_rate": round(conversion_rate, 4),
})
return pd.DataFrame(rows)

View File

@@ -0,0 +1,158 @@
"""
Shared, pure feature-engineering helpers.
Nothing in this module touches the database or the network - every
function takes plain Python / pandas structures in and returns them
out, which is what lets `tests/test_intelligence.py` exercise the real
ML feature logic without a live Postgres instance.
"""
from __future__ import annotations
import math
from datetime import date, datetime
from typing import Dict, Iterable, List, Optional
import numpy as np
import pandas as pd
# ---------------------------------------------------------------------------
# Category -> shelf-life bucket (used for expiry-aware discounting).
# Perishables get a short simulated shelf life; ambient/packaged goods get
# a long one (effectively "doesn't expire for pricing purposes").
# ---------------------------------------------------------------------------
PERISHABLE_SHELF_LIFE_DAYS = {
"dairy": 10,
"bakery & breads": 4,
"cakes & muffins": 5,
}
DEFAULT_SHELF_LIFE_DAYS = 270 # ambient FMCG (biscuits, tea, soap, ...)
STORE_TIER_ORDER = ["budget", "standard", "premium"]
def shelf_life_days_for_category(category: Optional[str]) -> int:
key = (category or "").strip().lower()
for k, v in PERISHABLE_SHELF_LIFE_DAYS.items():
if k in key:
return v
return DEFAULT_SHELF_LIFE_DAYS
def days_to_expiry(category: Optional[str], days_since_stocked: int) -> int:
"""Simulated remaining shelf life. Clamped at 0 (already expired stock
would have been written off, so this floors at 0 rather than going
negative)."""
life = shelf_life_days_for_category(category)
return max(0, life - int(days_since_stocked))
def cyclical_month_features(as_of: date) -> Dict[str, float]:
"""Sine/cosine encode the month so "December" and "January" are close
in feature space (seasonality wraps around the year)."""
angle = 2 * math.pi * (as_of.month - 1) / 12.0
return {"month_sin": math.sin(angle), "month_cos": math.cos(angle)}
def cyclical_dow_features(as_of: date) -> Dict[str, float]:
angle = 2 * math.pi * as_of.weekday() / 7.0
return {"dow_sin": math.sin(angle), "dow_cos": math.cos(angle)}
def is_festive_season(as_of: date) -> int:
"""Coarse Indian FMCG festive-demand window (Oct-Nov: Diwali season;
Aug: Independence Day/Onam-ish promo season). Used as a simple
seasonal-trend signal rather than hardcoding a discount bump - the
model learns how much this feature matters from the training data."""
return 1 if as_of.month in (10, 11) or as_of.month == 8 else 0
def stock_ratio(available_stock: int, reorder_level: int) -> float:
""">1 means well-stocked relative to reorder point; <1 means at/below
the reorder point. Reorder level is floored at 1 to avoid div-by-zero
for misconfigured rows."""
return float(available_stock) / float(max(reorder_level, 1))
def days_of_cover(available_stock: int, avg_daily_sales: float) -> float:
"""How many days current stock would last at the recent sales pace.
A very small floor on avg_daily_sales avoids an artificial 'infinite'
days-of-cover for a product that just hasn't sold yet."""
return float(available_stock) / max(float(avg_daily_sales), 0.05)
def sales_velocity(order_items: pd.DataFrame, store_id: str, brand: str, image_id: str,
as_of: date, window_days: int) -> Dict[str, float]:
"""Units/day and revenue/day sold for one (store, product) over the
trailing `window_days` ending at `as_of` (exclusive of future data -
callers must only pass order history up to `as_of`).
`order_items` is expected to have columns:
store_id, brand, image_id, order_date (datetime64), quantity, line_total
"""
if order_items.empty:
return {"units_per_day": 0.0, "revenue_per_day": 0.0, "order_count": 0}
window_start = pd.Timestamp(as_of) - pd.Timedelta(days=window_days)
mask = (
(order_items["store_id"] == store_id)
& (order_items["brand"] == brand)
& (order_items["image_id"] == image_id)
& (order_items["order_date"] >= window_start)
& (order_items["order_date"] < pd.Timestamp(as_of))
)
sub = order_items.loc[mask]
units = float(sub["quantity"].sum())
revenue = float(sub["line_total"].sum())
return {
"units_per_day": units / max(window_days, 1),
"revenue_per_day": revenue / max(window_days, 1),
"order_count": int(len(sub)),
}
def rfm_features(orders: pd.DataFrame, customer_id: str, as_of: date) -> Dict[str, float]:
"""Recency / Frequency / Monetary features for one customer, computed
only from orders strictly before `as_of` (so this is safe to use as a
training feature with a held-out future window as the label).
`orders` columns: customer_id, order_date (datetime64), order_value
"""
hist = orders[(orders["customer_id"] == customer_id) & (orders["order_date"] < pd.Timestamp(as_of))]
if hist.empty:
return {"recency_days": 999.0, "frequency": 0.0, "monetary": 0.0, "avg_order_value": 0.0}
last_order = hist["order_date"].max()
recency_days = (pd.Timestamp(as_of) - last_order).days
frequency = float(len(hist))
monetary = float(hist["order_value"].sum())
return {
"recency_days": float(recency_days),
"frequency": frequency,
"monetary": monetary,
"avg_order_value": monetary / frequency,
}
def normalize_0_100(series: pd.Series) -> pd.Series:
"""Min-max normalize a numeric series to a 0-100 range. Constant
series map to 50 (avoids div-by-zero and avoids an arbitrary 0)."""
lo, hi = series.min(), series.max()
if hi - lo < 1e-9:
return pd.Series([50.0] * len(series), index=series.index)
return (series - lo) / (hi - lo) * 100.0
def encode_category(categories: Iterable[str]) -> pd.Series:
"""Simple stable frequency-encoding for a category column - keeps
every model's category feature deterministic across train/inference
without needing to persist a fitted OneHotEncoder for a small
cardinality field."""
s = pd.Series(list(categories)).fillna("Uncategorized")
freq = s.value_counts(normalize=True)
return s.map(freq).fillna(0.0)
def encode_store_tier(tier: str) -> int:
try:
return STORE_TIER_ORDER.index((tier or "standard").lower())
except ValueError:
return 1

View File

@@ -0,0 +1,140 @@
"""
Feature 9: Demand Forecasting / Inventory Forecasting.
Approach: rolling-window time-series feature engineering (7/14/30-day
trailing rolling means of units sold, day-of-week and month cyclical
encoding, festive-season flag) feeding a RandomForestRegressor that
predicts expected average daily demand over the NEXT 7 days. This is
the same "engineer time features, then regress" pattern used for
trending (see `trending_model.py`), applied here to forecast forward
instead of score the present. Chosen over Prophet/LSTM for the
hardware-conscious reasons documented in `app/intelligence/__init__.py`.
Predicted daily demand also directly drives `available_stock -
predicted_demand * lead_time_days` for inventory forecasting, so one
model serves both "Demand Forecasting" and "Inventory Forecasting" in
Feature 9's suggested-models table rather than duplicating near-
identical logic in two places.
Pooled by (category, store_tier) rather than trained per exact product:
with a handful of stores and a simulated order history, most individual
products don't have enough daily data points for a standalone
time-series model to learn anything - pooling similar products' rolling
patterns gives the model enough signal while still producing a
per-product-per-store forecast at inference time (each row is scored
individually; only the training data is pooled).
"""
from __future__ import annotations
from datetime import date, timedelta
from typing import List
import numpy as np
import pandas as pd
from app.intelligence import features as F
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
MODEL_NAME = "demand_forecast_model"
FEATURE_COLUMNS = [
"rolling_mean_7", "rolling_mean_14", "rolling_mean_30", "month_sin", "month_cos",
"dow_sin", "dow_cos", "is_festive_season", "category_freq", "store_tier_encoded",
]
def _daily_series(order_items: pd.DataFrame, store_id: str, brand: str, image_id: str) -> pd.Series:
sub = order_items[
(order_items["store_id"] == store_id) & (order_items["brand"] == brand) & (order_items["image_id"] == image_id)
]
if sub.empty:
return pd.Series(dtype=float)
daily = sub.groupby(sub["order_date"].dt.date)["quantity"].sum()
daily.index = pd.to_datetime(daily.index)
return daily.asfreq("D", fill_value=0)
def build_training_frame(
order_items: pd.DataFrame,
product_meta: pd.DataFrame, # columns: store_id, brand, image_id, category, store_tier
as_of_dates: List[date],
) -> pd.DataFrame:
"""For each (store, product, as_of) sample a rolling-feature row and
the REAL (not synthetic) label: actual average daily units sold in
the 7 days AFTER as_of. This is genuine supervised time-series
forecasting - the label comes straight from the simulated ground
truth, no bootstrap formula needed here (unlike discount/trending)."""
cat_freq = F.encode_category(product_meta["category"])
rows = []
for i, meta in product_meta.reset_index(drop=True).iterrows():
series = _daily_series(order_items, meta["store_id"], meta["brand"], meta["image_id"])
if series.empty:
continue
for as_of in as_of_dates:
as_of_ts = pd.Timestamp(as_of)
history = series[series.index < as_of_ts]
future = series[(series.index >= as_of_ts) & (series.index < as_of_ts + pd.Timedelta(days=7))]
if len(history) < 14 or future.empty:
continue
row = {
"rolling_mean_7": history.tail(7).mean(),
"rolling_mean_14": history.tail(14).mean(),
"rolling_mean_30": history.tail(30).mean(),
**F.cyclical_month_features(as_of),
**F.cyclical_dow_features(as_of),
"is_festive_season": F.is_festive_season(as_of),
"category_freq": cat_freq.iloc[i],
"store_tier_encoded": F.encode_store_tier(meta["store_tier"]),
"target_avg_daily_units": future.mean(),
}
rows.append(row)
return pd.DataFrame(rows)
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.ensemble import RandomForestRegressor
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_absolute_error
X = training_frame[FEATURE_COLUMNS]
y = training_frame["target_avg_daily_units"]
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
model = RandomForestRegressor(n_estimators=200, max_depth=8, min_samples_leaf=3, random_state=42, n_jobs=-1)
model.fit(X_train, y_train)
mae = float(mean_absolute_error(y_test, model.predict(X_test)))
bundle = ModelBundle(
estimator=model, feature_columns=FEATURE_COLUMNS, model_name=MODEL_NAME,
n_samples=len(training_frame), extra={"val_mae_units_per_day": round(mae, 3)},
)
save_bundle(bundle)
return bundle
class DemandForecaster:
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def forecast(self, feature_df: pd.DataFrame, horizon_days: int = 7) -> pd.DataFrame:
"""Returns feature_df with `forecast_avg_daily_units` and
`forecast_total_units` (over horizon_days) columns added. Falls
back to the naive rolling_mean_7 (still real data, just not
model-refined) if no model is trained yet - never a fixed
constant."""
out = feature_df.copy()
if not self._ensure_loaded() or feature_df.empty:
out["forecast_avg_daily_units"] = out.get("rolling_mean_7", 0.0)
else:
X = feature_df[self._bundle.feature_columns]
out["forecast_avg_daily_units"] = np.clip(self._bundle.estimator.predict(X), 0, None)
out["forecast_total_units"] = out["forecast_avg_daily_units"] * horizon_days
return out
demand_forecaster = DemandForecaster()

View File

@@ -0,0 +1,61 @@
"""Shared persistence helper for every trained model in this package.
Keeping load/save in one place (rather than duplicated per model file)
is the SOLID/DRY-motivated reason this exists - every *_model.py file
just calls `save_bundle` / `load_bundle` with its own feature list and
estimator.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional
logger = logging.getLogger(__name__)
ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts"
ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True)
@dataclass
class ModelBundle:
"""Everything needed to reproduce a prediction: the fitted estimator,
the exact feature column order it expects, and light metadata for
the admin/health endpoints to report on (trained_at, n_samples,
version)."""
estimator: Any
feature_columns: List[str]
model_name: str
version: str = "v1"
trained_at: str = field(default_factory=lambda: datetime.now(timezone.utc).isoformat())
n_samples: int = 0
extra: Dict[str, Any] = field(default_factory=dict)
def artifact_path(model_name: str) -> Path:
return ARTIFACTS_DIR / f"{model_name}.joblib"
def save_bundle(bundle: ModelBundle) -> Path:
import joblib # lazy import: keeps API boot fast if scikit-learn isn't needed yet
path = artifact_path(bundle.model_name)
joblib.dump(bundle, path)
logger.info("Saved model bundle '%s' (%d samples) -> %s", bundle.model_name, bundle.n_samples, path)
return path
def load_bundle(model_name: str) -> Optional[ModelBundle]:
import joblib
path = artifact_path(model_name)
if not path.exists():
logger.warning("No trained model artifact found for '%s' at %s - run scripts/train_ml_models.py first", model_name, path)
return None
try:
return joblib.load(path)
except Exception as e: # noqa: BLE001
logger.error("Failed to load model bundle '%s': %s", model_name, e)
return None

View File

@@ -0,0 +1,115 @@
"""
Feature 14: "Nutrition-Based Clustering".
Groups products into nutrition-profile clusters with scikit-learn's
KMeans over the same normalized feature space as the similarity model.
Cluster labels are derived transparently from each cluster's own
centroid statistics (e.g. "High Protein / Low Sugar") rather than
LLM-named, so a cluster's name is always traceable back to real
aggregate numbers.
Used for: the `nutrition_cluster` / `nutrition_cluster_label` columns
on `nutrition_insights` (surfaced in the product detail view and the
analytics dashboard), and as a candidate pool for
`nutrition_alternatives_service.py` (restricting "healthier
alternative" search to a nutritionally-similar cluster rather than the
whole catalog).
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Tuple
import numpy as np
import pandas as pd
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
from app.intelligence.nutrition_similarity import FEATURE_COLUMNS, MIN_VERIFIED_FRACTION
logger = logging.getLogger(__name__)
MODEL_NAME = "nutrition_clustering"
DEFAULT_N_CLUSTERS = 6
def _label_cluster(centroid: pd.Series, overall_median: pd.Series) -> str:
"""Names a cluster from how its centroid compares to the whole
catalog's median on the two or three most distinctive nutrients -
entirely derived from the data, not hand-authored per cluster."""
descriptors: List[Tuple[str, float]] = []
readable = {
"protein_g": "High Protein", "dietary_fiber_g": "High Fiber",
"total_sugar_g": "High Sugar", "sodium_mg": "High Sodium",
"saturated_fat_g": "High Saturated Fat", "calories_kcal": "Calorie Dense",
}
for col, label in readable.items():
if col not in centroid or col not in overall_median or overall_median[col] in (0, None):
continue
ratio = centroid[col] / overall_median[col] if overall_median[col] else 1.0
if ratio >= 1.3:
descriptors.append((label, ratio))
elif ratio <= 0.7 and ratio > 0 and col in ("total_sugar_g", "sodium_mg", "saturated_fat_g", "calories_kcal"):
descriptors.append((f"Low {label.replace('High ', '')}", 1 / ratio))
descriptors.sort(key=lambda t: t[1], reverse=True)
top = [d[0] for d in descriptors[:2]]
return " / ".join(top) if top else "Balanced Profile"
def train_clusters(df: pd.DataFrame, n_clusters: int = DEFAULT_N_CLUSTERS) -> Dict[str, Any]:
from sklearn.cluster import KMeans
from sklearn.preprocessing import StandardScaler
if df.empty:
return {"trained": False, "reason": "no verified nutrition rows"}
df = df.copy()
verified_fraction = df[FEATURE_COLUMNS].notna().sum(axis=1) / len(FEATURE_COLUMNS)
df = df[verified_fraction >= MIN_VERIFIED_FRACTION].reset_index(drop=True)
k = min(n_clusters, max(2, len(df) // 3))
if len(df) < k * 2:
return {"trained": False, "reason": f"only {len(df)} products have enough verified fields for {k} clusters"}
# Median-impute missing values; if a whole column is missing (median is
# NaN), fall back to 0.0 so KMeans never receives NaN.
matrix = df[FEATURE_COLUMNS].apply(lambda col: col.fillna(col.median()).fillna(0.0))
scaler = StandardScaler()
scaled = scaler.fit_transform(matrix.values)
km = KMeans(n_clusters=k, n_init=10, random_state=42)
labels = km.fit_predict(scaled)
overall_median = matrix.median()
cluster_names: Dict[int, str] = {}
for c in range(k):
centroid_raw = matrix[labels == c].median()
cluster_names[c] = _label_cluster(centroid_raw, overall_median)
assignments = {
(brand, image_id): {"cluster": int(c), "label": cluster_names[int(c)]}
for brand, image_id, c in zip(df["brand"], df["image_id"], labels)
}
bundle = ModelBundle(
estimator=km,
feature_columns=FEATURE_COLUMNS,
model_name=MODEL_NAME,
n_samples=len(df),
extra={"scaler": scaler, "cluster_names": cluster_names, "assignments": assignments},
)
save_bundle(bundle)
logger.info(f"Trained nutrition clustering (k={k}) on {len(df)} products")
return {"trained": True, "n_samples": len(df), "n_clusters": k, "cluster_labels": cluster_names}
def get_assignments() -> Dict[Tuple[str, str], Dict[str, Any]]:
bundle = load_bundle(MODEL_NAME)
if not bundle:
return {}
return bundle.extra.get("assignments", {})
def get_cluster_members(cluster: int) -> List[Tuple[str, str]]:
bundle = load_bundle(MODEL_NAME)
if not bundle:
return []
return [key for key, info in bundle.extra.get("assignments", {}).items() if info["cluster"] == cluster]

View File

@@ -0,0 +1,143 @@
"""
Feature 10: "Personalized Nutrition Recommendations".
Builds each customer's nutrient-purchase profile from the existing
Store Intelligence `orders`/`order_items` tables (v3.0 layer - see
[[logistics-ml-pipelines]] history) joined against verified
`nutrition_facts`, then applies the three example behaviors from the
spec directly:
- frequently buys high-protein items -> recommend more high-protein items
- frequently buys high-sugar snacks -> recommend healthier alternatives
- frequently buys low-fat items -> recommend similar low-fat items
This is content-based filtering over a nutrition feature space (the
"Content-Based Filtering" + "Nutrition Similarity" options from the
spec's algorithm list) rather than collaborative filtering, since it
needs to work for a single customer's history without requiring
enough cross-customer overlap to train a collaborative model - a
reasonable simplification given the 8GB RAM / CPU-only environment and
the size of a simulated order dataset.
Gracefully returns [] if the Store Intelligence order tables haven't
been seeded - this is an optional enhancement on top of that layer,
not a hard dependency of the nutrition module.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List
from app.services import nutrition_db, nutrition_scoring
from app.services.nutrition_alternatives_service import find_alternatives
from app.services.vector_store import _connect
logger = logging.getLogger(__name__)
HIGH_SUGAR_PURCHASE_THRESHOLD = nutrition_scoring.THRESHOLDS["sugar_high"]
HIGH_PROTEIN_PURCHASE_THRESHOLD = nutrition_scoring.THRESHOLDS["protein_high_g"]
LOW_FAT_PURCHASE_THRESHOLD = nutrition_scoring.THRESHOLDS["low_fat_ceiling"]
def _customer_purchase_profile(customer_id: str) -> List[Dict[str, Any]]:
"""Every (brand, image_id) the customer has ordered, with verified
nutrition facts attached, weighted by how many times they bought it."""
conn = _connect()
if not conn:
return []
try:
from psycopg.rows import dict_row
with conn, conn.cursor(row_factory=dict_row) as cur:
cur.execute(
"SELECT EXISTS (SELECT FROM information_schema.tables WHERE table_name = 'order_items')"
)
if not cur.fetchone()["exists"]:
return []
cur.execute(
"""
SELECT oi.brand, oi.image_id, SUM(oi.quantity) AS times_purchased,
f.category, f.protein_g, f.total_sugar_g, f.total_fat_g,
f.dietary_fiber_g, i.health_score
FROM order_items oi
JOIN orders o ON o.order_id = oi.order_id
LEFT JOIN nutrition_facts f ON f.brand = oi.brand AND f.image_id = oi.image_id
LEFT JOIN nutrition_insights i ON i.brand = oi.brand AND i.image_id = oi.image_id
WHERE o.customer_id = %s
GROUP BY oi.brand, oi.image_id, f.category, f.protein_g, f.total_sugar_g,
f.total_fat_g, f.dietary_fiber_g, i.health_score
""",
(customer_id,),
)
return [dict(r) for r in cur.fetchall()]
except Exception as e: # noqa: BLE001
logger.error(f"_customer_purchase_profile failed for {customer_id}: {e}")
return []
finally:
conn.close()
def recommend_for_customer(customer_id: str, top_k: int = 8) -> Dict[str, Any]:
purchases = _customer_purchase_profile(customer_id)
scored_purchases = [p for p in purchases if p.get("protein_g") is not None or p.get("total_sugar_g") is not None]
if not scored_purchases:
return {"customer_id": customer_id, "purchase_pattern": "insufficient_data", "recommendations": []}
def weighted_avg(field: str) -> float:
vals = [(p[field], p["times_purchased"]) for p in scored_purchases if p.get(field) is not None]
if not vals:
return 0.0
total_weight = sum(w for _, w in vals)
return sum(v * w for v, w in vals) / total_weight if total_weight else 0.0
avg_protein = weighted_avg("protein_g")
avg_sugar = weighted_avg("total_sugar_g")
avg_fat = weighted_avg("total_fat_g")
purchased_keys = {(p["brand"], p["image_id"]) for p in purchases}
recommendations: List[Dict[str, Any]] = []
pattern: str
if avg_sugar >= HIGH_SUGAR_PURCHASE_THRESHOLD:
pattern = "frequent_high_sugar_purchases"
# Pull healthier alternatives around their most-purchased high-sugar item.
worst = max(
(p for p in scored_purchases if p.get("total_sugar_g") is not None),
key=lambda p: (p["total_sugar_g"], p["times_purchased"]),
default=None,
)
if worst:
alts = find_alternatives(worst["brand"], worst["image_id"], top_k=top_k)
recommendations = [{**a, "recommendation_reason": "Healthier alternative to a frequently purchased high-sugar item"} for a in alts]
elif avg_protein >= HIGH_PROTEIN_PURCHASE_THRESHOLD:
pattern = "frequent_high_protein_purchases"
results = nutrition_db.query_products(sort_by="protein", order="desc", diet_tag="High Protein", limit=30)
recommendations = [
{**r, "recommendation_reason": "Matches your frequent high-protein purchases"}
for r in results if (r["brand"], r["image_id"]) not in purchased_keys
][:top_k]
elif avg_fat > 0 and avg_fat <= LOW_FAT_PURCHASE_THRESHOLD:
pattern = "frequent_low_fat_purchases"
results = nutrition_db.query_products(sort_by="health_score", order="desc", diet_tag="Low Fat", limit=30)
recommendations = [
{**r, "recommendation_reason": "Similar low-fat profile to your recent purchases"}
for r in results if (r["brand"], r["image_id"]) not in purchased_keys
][:top_k]
else:
pattern = "general"
results = nutrition_db.query_products(sort_by="health_score", order="desc", limit=30)
recommendations = [
{**r, "recommendation_reason": "Highly rated for overall nutrition"}
for r in results if (r["brand"], r["image_id"]) not in purchased_keys
][:top_k]
return {
"customer_id": customer_id,
"purchase_pattern": pattern,
"avg_protein_g": round(avg_protein, 1),
"avg_sugar_g": round(avg_sugar, 1),
"avg_fat_g": round(avg_fat, 1),
"recommendations": recommendations,
}

View File

@@ -0,0 +1,113 @@
"""
Feature 8: nutritional similarity, ML-based.
Uses scikit-learn's `NearestNeighbors` with cosine distance over a
normalized nutrient-vector feature space (protein, calories, fiber,
sugar, fat, sodium, + core micronutrients where available) - satisfying
the spec's explicit "Cosine Similarity" and "KNN" options while staying
inside the 8GB RAM / CPU-only budget documented for this environment
(no embeddings/sentence-transformers needed for ~a dozen numeric
features; that would be over-engineering for this feature).
IMPORTANT SCOPE NOTE: missing nutrient values are median-imputed *only*
inside this in-memory feature matrix, purely so the distance metric is
computable. This never writes an imputed number back into
`nutrition_facts` - the database only ever holds verified values (see
`nutrition_data_service.py`). Imputation here is a standard ML
pre-processing step for the similarity model, not a claim about any
product's actual nutrition.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List
import numpy as np
import pandas as pd
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
logger = logging.getLogger(__name__)
MODEL_NAME = "nutrition_similarity"
FEATURE_COLUMNS = [
"calories_kcal", "protein_g", "carbohydrates_g", "dietary_fiber_g",
"total_sugar_g", "total_fat_g", "saturated_fat_g", "sodium_mg",
"calcium_mg", "iron_mg", "vitamin_c_mg", "potassium_mg",
]
# Minimum fraction of the feature columns that must be verified (non-null
# before imputation) for a product to be included in the similarity
# index at all - keeps products with almost no real data out of the
# comparison space entirely rather than comparing on mostly-imputed noise.
MIN_VERIFIED_FRACTION = 0.35
def train_similarity_index(df: pd.DataFrame) -> Dict[str, Any]:
"""`df` is `nutrition_db.get_all_nutrition_facts_df()`. Fits a
StandardScaler + NearestNeighbors bundle and persists it via the
shared model_utils pattern."""
from sklearn.neighbors import NearestNeighbors
from sklearn.preprocessing import StandardScaler
if df.empty:
return {"trained": False, "reason": "no verified nutrition rows"}
df = df.copy()
verified_fraction = df[FEATURE_COLUMNS].notna().sum(axis=1) / len(FEATURE_COLUMNS)
df = df[verified_fraction >= MIN_VERIFIED_FRACTION].reset_index(drop=True)
if len(df) < 3:
return {"trained": False, "reason": f"only {len(df)} products have enough verified fields (need >= 3)"}
# Median-impute missing values. If an entire column is missing (median
# itself is NaN - happens when no product has that nutrient verified),
# fall back to 0.0 so no NaN ever reaches the scaler / NearestNeighbors.
matrix = df[FEATURE_COLUMNS].apply(lambda col: col.fillna(col.median()).fillna(0.0))
scaler = StandardScaler()
scaled = scaler.fit_transform(matrix.values)
n_neighbors = min(11, len(df)) # self + up to 10 neighbors
nn = NearestNeighbors(n_neighbors=n_neighbors, metric="cosine")
nn.fit(scaled)
bundle = ModelBundle(
estimator=nn,
feature_columns=FEATURE_COLUMNS,
model_name=MODEL_NAME,
n_samples=len(df),
extra={
"scaler": scaler,
"product_keys": list(zip(df["brand"], df["image_id"])),
},
)
save_bundle(bundle)
logger.info(f"Trained nutrition similarity index on {len(df)} products")
return {"trained": True, "n_samples": len(df)}
def find_similar(brand: str, image_id: str, top_k: int = 5) -> List[Dict[str, Any]]:
bundle = load_bundle(MODEL_NAME)
if not bundle:
return []
keys: List[tuple] = bundle.extra["product_keys"]
try:
idx = keys.index((brand, image_id))
except ValueError:
return [] # product wasn't in the trained index (too little verified data, or trained before it was enriched)
nn = bundle.estimator
scaler = bundle.extra["scaler"]
query_vec = nn._fit_X[idx].reshape(1, -1) # already-scaled training vector, avoids re-scaling drift
distances, indices = nn.kneighbors(query_vec, n_neighbors=min(top_k + 1, len(keys)))
results = []
for dist, i in zip(distances[0], indices[0]):
cand_brand, cand_image_id = keys[i]
if cand_brand == brand and cand_image_id == image_id:
continue
similarity = round(max(0.0, 1.0 - float(dist)), 4) # cosine distance -> similarity
results.append({"brand": cand_brand, "image_id": cand_image_id, "similarity_score": similarity})
if len(results) >= top_k:
break
return results

View File

@@ -0,0 +1,186 @@
"""
Synthetic order-history generator.
Generates a realistic-looking transaction log for the simulated 5-store
retail environment: which customer bought what, from which store, on
which day, for how much, via which payment method, with what delivery
outcome. This is the ground truth that Feature 8 (Order Simulation)
asks for, and it is also the raw material every other intelligent
feature is trained/computed from:
- Trending detection reads recent order velocity per product.
- The recommendation engine's collaborative-filtering signal reads
which products co-occur in the same order.
- The discount model's demand/velocity features come from here.
- Store/product analytics (revenue, profit, basket value, footfall)
are aggregated directly from this data.
- Purchase-propensity classification uses real (not synthetic-formula)
labels derived from a time-split of this data.
Everything here is deterministic given a seed, so re-running the seed
script reproduces the same catalog/order history - important for
repeatable ML training and for demos.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from datetime import date, timedelta
from typing import Dict, List, Optional, Sequence
import hashlib
import numpy as np
import pandas as pd
PAYMENT_METHODS = ["UPI", "Credit/Debit Card", "Cash on Delivery", "Wallet", "Net Banking"]
PAYMENT_WEIGHTS = [0.46, 0.20, 0.18, 0.11, 0.05] # India-skewed toward UPI
DELIVERY_STATUSES = ["Delivered", "Delivered", "Delivered", "Delivered", "Pending", "Cancelled", "Returned"]
STORE_TIER_DEMAND_MULTIPLIER = {"budget": 0.85, "standard": 1.0, "premium": 1.25}
@dataclass
class StoreProfile:
store_id: str
store_name: str
city: str
tier: str # "budget" | "standard" | "premium"
footfall_index: float # baseline avg orders/day before weekday/seasonal effects
@dataclass
class StoreCatalogEntry:
store_id: str
brand: str
image_id: str
category: str
selling_price: float
available_stock: int
def deterministic_unit(seed_key: str) -> float:
"""Same helper pattern as price_estimator._deterministic_unit: a
stable pseudo-random value in [0, 1] from a string seed, so latent
popularity is reproducible across runs without persisting it."""
digest = hashlib.md5(seed_key.strip().lower().encode("utf-8")).hexdigest()
return int(digest[:8], 16) / 0xFFFFFFFF
def latent_popularity(brand: str, image_id: str) -> float:
"""A hidden 0.2-1.0 'true popularity' per product, used only to bias
which products get ordered more often during simulation - this is
what gives the trending/popularity models a real signal to recover
from the resulting order data, rather than every product selling at
a uniform random rate."""
u = deterministic_unit(f"popularity|{brand}|{image_id}")
# Skew toward a Pareto-ish long tail: most products are middling,
# a minority are hits - matches real retail sales distribution.
return 0.2 + (u ** 2.2) * 0.8
def _weekday_multiplier(d: date) -> float:
# Fri/Sat/Sun busier than midweek.
return {0: 0.9, 1: 0.9, 2: 0.95, 3: 1.0, 4: 1.15, 5: 1.35, 6: 1.2}[d.weekday()]
def _festive_multiplier(d: date) -> float:
if d.month in (10, 11): # Diwali season
return 1.4
if d.month == 8: # Independence Day / monsoon promo season
return 1.15
return 1.0
def simulate_orders(
stores: Sequence[StoreProfile],
store_catalogs: Dict[str, List[StoreCatalogEntry]],
start_date: date,
end_date: date,
seed: int = 42,
customers_per_store: int = 220,
) -> "SimulationResult":
"""Generate `orders` and `order_items` DataFrames covering
[start_date, end_date] inclusive, for every store.
Returns a SimulationResult with two DataFrames ready to persist or
feed straight into feature engineering / model training.
"""
rng = np.random.default_rng(seed)
order_rows: List[dict] = []
item_rows: List[dict] = []
order_seq = 0
for store in stores:
catalog = store_catalogs.get(store.store_id, [])
if not catalog:
continue
weights = np.array([latent_popularity(e.brand, e.image_id) for e in catalog])
weights = weights / weights.sum()
customer_ids = [f"CUST-{store.store_id}-{i:04d}" for i in range(customers_per_store)]
# A minority of "regular" customers order much more often than
# the rest, which is what gives RFM/purchase-propensity features
# something meaningful to learn from.
customer_affinity = rng.pareto(a=2.2, size=len(customer_ids)) + 0.15
tier_mult = STORE_TIER_DEMAND_MULTIPLIER.get(store.tier, 1.0)
current = start_date
while current <= end_date:
lam = store.footfall_index * tier_mult * _weekday_multiplier(current) * _festive_multiplier(current)
n_orders_today = rng.poisson(lam=max(lam, 0.1))
if n_orders_today > 0:
cust_p = customer_affinity / customer_affinity.sum()
todays_customers = rng.choice(customer_ids, size=n_orders_today, p=cust_p)
for cust_id in todays_customers:
order_seq += 1
order_id = f"ORD-{store.store_id}-{order_seq:07d}"
n_items = int(rng.integers(1, 5))
picks = rng.choice(len(catalog), size=min(n_items, len(catalog)), replace=False, p=weights)
order_total = 0.0
for idx in picks:
entry = catalog[idx]
qty = int(rng.integers(1, 4))
line_total = round(entry.selling_price * qty, 2)
order_total += line_total
item_rows.append({
"order_id": order_id,
"store_id": store.store_id,
"brand": entry.brand,
"image_id": entry.image_id,
"quantity": qty,
"unit_price": entry.selling_price,
"line_total": line_total,
})
payment = rng.choice(PAYMENT_METHODS, p=PAYMENT_WEIGHTS)
status = rng.choice(DELIVERY_STATUSES)
order_rows.append({
"order_id": order_id,
"customer_id": cust_id,
"store_id": store.store_id,
"order_date": pd.Timestamp(current),
"payment_method": payment,
"order_value": round(order_total, 2),
"delivery_status": status,
})
current += timedelta(days=1)
orders_df = pd.DataFrame(order_rows)
items_df = pd.DataFrame(item_rows)
return SimulationResult(orders=orders_df, order_items=items_df)
@dataclass
class SimulationResult:
orders: pd.DataFrame
order_items: pd.DataFrame
def summary(self) -> dict:
if self.orders.empty:
return {"total_orders": 0, "total_revenue": 0.0, "date_range": None}
return {
"total_orders": int(len(self.orders)),
"total_order_items": int(len(self.order_items)),
"total_revenue": float(self.orders["order_value"].sum()),
"date_range": [
str(self.orders["order_date"].min().date()),
str(self.orders["order_date"].max().date()),
],
}

View File

@@ -0,0 +1,82 @@
"""
Feature 9 / Feature 5: Popularity Prediction (Regression).
Blends simulated engagement signals (views, wishlist adds - see
`app/intelligence/engagement_simulation.py`) with real simulated
purchase behaviour (orders, conversion rate) and a simulated rating,
into a single 0-100 popularity score. Bootstrapped the same way as the
discount model (see `synthetic_labels.py`): a documented weighted
formula generates training labels, a RandomForestRegressor learns the
general relationship, and only the trained model is used for serving -
which matters once real telemetry replaces the simulated
views/wishlist/rating inputs (the model doesn't need to change, only
its training data source does).
"""
from __future__ import annotations
import numpy as np
import pandas as pd
from app.intelligence import features as F
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
from app.intelligence.synthetic_labels import synthetic_popularity_score
MODEL_NAME = "popularity_model"
FEATURE_COLUMNS = ["views_norm", "wishlist_norm", "orders_norm", "rating_norm", "conversion_norm"]
def build_training_frame(product_engagement: pd.DataFrame) -> pd.DataFrame:
"""`product_engagement` columns: views, wishlist_count, orders_count,
avg_rating (1-5), conversion_rate (0-1)."""
df = pd.DataFrame({
"views_norm": F.normalize_0_100(product_engagement["views"]),
"wishlist_norm": F.normalize_0_100(product_engagement["wishlist_count"]),
"orders_norm": F.normalize_0_100(product_engagement["orders_count"]),
"rating_norm": F.normalize_0_100(product_engagement["avg_rating"]),
"conversion_norm": F.normalize_0_100(product_engagement["conversion_rate"]),
})
rng = np.random.default_rng(23)
df["popularity_score"] = synthetic_popularity_score(df, rng)
return df
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.ensemble import RandomForestRegressor
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_absolute_error
X = training_frame[FEATURE_COLUMNS]
y = training_frame["popularity_score"]
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
model = RandomForestRegressor(n_estimators=150, max_depth=6, random_state=42, n_jobs=-1)
model.fit(X_train, y_train)
mae = float(mean_absolute_error(y_test, model.predict(X_test)))
bundle = ModelBundle(
estimator=model, feature_columns=FEATURE_COLUMNS, model_name=MODEL_NAME,
n_samples=len(training_frame), extra={"val_mae_points": round(mae, 3)},
)
save_bundle(bundle)
return bundle
class PopularityScorer:
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def score(self, feature_df: pd.DataFrame) -> pd.Series:
if not self._ensure_loaded() or feature_df.empty:
return pd.Series([0.0] * len(feature_df), index=feature_df.index)
X = feature_df[self._bundle.feature_columns]
preds = np.clip(self._bundle.estimator.predict(X), 0, 100)
return pd.Series(preds, index=feature_df.index)
popularity_scorer = PopularityScorer()

View File

@@ -0,0 +1,99 @@
"""
Feature 9: Customer Purchase Prediction (Classification).
Binary classifier: will this customer place another order in the next
14-day window, given their RFM (Recency/Frequency/Monetary) history up
to a cut-off date? Unlike the discount/trending/popularity models, this
one needs NO synthetic label bootstrap - the label is real: we
time-split the simulated order history at a cut-off date, compute RFM
features from everything before it, and label = 1 if that customer has
>=1 order in the 14 days after it, else 0. This is standard churn/
purchase-propensity modelling methodology applied to (simulated) real
transactions.
LogisticRegression: a classification task with well-behaved, roughly
linearly-separable RFM features doesn't need a heavier model, and it
gives directly interpretable coefficients (useful for a "why" behind a
propensity score in the analytics UI).
"""
from __future__ import annotations
from datetime import date, timedelta
from typing import List
import numpy as np
import pandas as pd
from app.intelligence import features as F
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
MODEL_NAME = "purchase_propensity_model"
FEATURE_COLUMNS = ["recency_days", "frequency", "monetary", "avg_order_value"]
PREDICTION_WINDOW_DAYS = 14
def build_training_frame(orders: pd.DataFrame, cutoff_dates: List[date]) -> pd.DataFrame:
"""`orders` columns: customer_id, order_date, order_value."""
rows = []
for cutoff in cutoff_dates:
cutoff_ts = pd.Timestamp(cutoff)
window_end = cutoff_ts + pd.Timedelta(days=PREDICTION_WINDOW_DAYS)
customers = orders.loc[orders["order_date"] < cutoff_ts, "customer_id"].unique()
future_buyers = set(
orders.loc[(orders["order_date"] >= cutoff_ts) & (orders["order_date"] < window_end), "customer_id"]
)
for cust in customers:
rfm = F.rfm_features(orders, cust, cutoff.__class__(cutoff.year, cutoff.month, cutoff.day))
rows.append({**rfm, "will_purchase": int(cust in future_buyers)})
return pd.DataFrame(rows)
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.linear_model import LogisticRegression
from sklearn.model_selection import train_test_split
from sklearn.metrics import roc_auc_score
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline
X = training_frame[FEATURE_COLUMNS]
y = training_frame["will_purchase"]
pipeline = Pipeline([("scale", StandardScaler()), ("clf", LogisticRegression(max_iter=500, class_weight="balanced"))])
auc = None
if y.nunique() > 1 and len(training_frame) >= 20:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)
pipeline.fit(X_train, y_train)
try:
auc = float(roc_auc_score(y_test, pipeline.predict_proba(X_test)[:, 1]))
except ValueError:
auc = None
else:
pipeline.fit(X, y)
bundle = ModelBundle(
estimator=pipeline, feature_columns=FEATURE_COLUMNS, model_name=MODEL_NAME,
n_samples=len(training_frame), extra={"val_auc": round(auc, 3) if auc is not None else None},
)
save_bundle(bundle)
return bundle
class PurchasePropensityPredictor:
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def predict_proba(self, feature_df: pd.DataFrame) -> pd.Series:
if not self._ensure_loaded() or feature_df.empty:
return pd.Series([0.5] * len(feature_df), index=feature_df.index)
X = feature_df[self._bundle.feature_columns]
proba = self._bundle.estimator.predict_proba(X)[:, 1]
return pd.Series(proba, index=feature_df.index)
purchase_propensity_predictor = PurchasePropensityPredictor()

View File

@@ -0,0 +1,151 @@
"""
Feature 7: ML-Based Product Recommendation Engine.
Three signals, blended (hybrid recommendation):
1. Embedding similarity (content-based, semantic). Reuses the SAME
sentence-transformers/all-MiniLM-L6-v2 embeddings the RAG pipeline
already computes and stores in pgvector for every product
(`app/services/embeddings_service.py`) - no new embedding model, no
extra inference cost. This is why "Tata Tea Gold" naturally recommends
"Brooke Bond Red Label" / "Taj Mahal Tea": their generated
descriptions land close together in embedding space regardless of
brand table.
2. TF-IDF similarity (content-based, lexical). A lightweight
scikit-learn TfidfVectorizer over title+category+brand text, added
because embedding similarity alone can miss exact-category/near-
duplicate matches when descriptions are stylistically different but
the products are practically identical substitutes (e.g. "Noodles"
across brands) - TF-IDF picks up shared category/brand vocabulary
that a semantic embedding sometimes smooths over.
3. Collaborative filtering (behavioural). Item-item cosine similarity
over a customer x product co-purchase matrix built from the
simulated order history (`order_items`) - "customers who bought X
also bought Y". Pure scipy/pandas, no extra ML library needed.
The final score is a weighted blend, and every recommendation returned
carries its own `similarity_score` (0-1) as the spec requires.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Dict, List, Optional, Sequence
import numpy as np
import pandas as pd
from scipy import sparse
DEFAULT_WEIGHTS = {"embedding": 0.45, "tfidf": 0.20, "collaborative": 0.20, "popularity": 0.15}
def cosine_sim_matrix(vectors: np.ndarray) -> np.ndarray:
"""Row-normalized cosine similarity matrix for a (n, dim) array."""
norms = np.linalg.norm(vectors, axis=1, keepdims=True)
norms[norms == 0] = 1e-9
normalized = vectors / norms
return normalized @ normalized.T
def embedding_similarity_to_source(source_vec: np.ndarray, candidate_vecs: np.ndarray) -> np.ndarray:
"""Cosine similarity of every row in candidate_vecs to a single
source_vec, mapped from pgvector's cosine *distance* convention
(0=identical, 2=opposite) is NOT used here - this takes raw
embedding vectors and computes similarity directly (1=identical,
-1=opposite), so callers passing pgvector distances must convert
first (`1 - distance` for pgvector's cosine distance)."""
source_norm = source_vec / max(np.linalg.norm(source_vec), 1e-9)
cand_norms = np.linalg.norm(candidate_vecs, axis=1)
cand_norms[cand_norms == 0] = 1e-9
return (candidate_vecs @ source_norm) / cand_norms
def tfidf_similarity(corpus: Sequence[str], source_index: int) -> np.ndarray:
"""TF-IDF cosine similarity of every document in `corpus` to
`corpus[source_index]`. `corpus` should be short text like
"<title> <category> <brand>" for each candidate product, with the
source product included at `source_index`."""
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.metrics.pairwise import cosine_similarity
if len(corpus) < 2:
return np.zeros(len(corpus))
vectorizer = TfidfVectorizer(stop_words="english", max_features=2000)
matrix = vectorizer.fit_transform(corpus)
sims = cosine_similarity(matrix[source_index], matrix).ravel()
return sims
def build_copurchase_matrix(order_items: pd.DataFrame) -> tuple[sparse.csr_matrix, List[str]]:
"""Builds a (n_products x n_products) item-item co-occurrence-based
cosine similarity matrix from order history. Products are keyed by
'brand||image_id'. Returns (similarity_matrix, product_key_index).
"""
if order_items.empty:
return sparse.csr_matrix((0, 0)), []
items = order_items.copy()
items["product_key"] = items["brand"] + "||" + items["image_id"]
product_keys = sorted(items["product_key"].unique())
key_to_idx = {k: i for i, k in enumerate(product_keys)}
order_ids = sorted(items["order_id"].unique())
order_to_idx = {o: i for i, o in enumerate(order_ids)}
rows = items["product_key"].map(key_to_idx).to_numpy()
cols = items["order_id"].map(order_to_idx).to_numpy()
data = np.ones(len(items))
basket_matrix = sparse.csr_matrix((data, (rows, cols)), shape=(len(product_keys), len(order_ids)))
# Item-item cosine similarity via normalized dot product of the
# (product x order) incidence matrix - standard, lightweight
# collaborative-filtering approach (no external CF library needed).
norms = np.sqrt(basket_matrix.multiply(basket_matrix).sum(axis=1)).A.ravel()
norms[norms == 0] = 1e-9
inv_norm = sparse.diags(1.0 / norms)
normalized = inv_norm @ basket_matrix
sim = normalized @ normalized.T
return sparse.csr_matrix(sim), product_keys
@dataclass
class RecommendationCandidate:
brand: str
image_id: str
embedding_similarity: float = 0.0
tfidf_similarity: float = 0.0
collaborative_similarity: float = 0.0
popularity_norm: float = 0.0 # 0-1
def hybrid_score(self, weights: Optional[Dict[str, float]] = None) -> float:
w = weights or DEFAULT_WEIGHTS
score = (
w["embedding"] * self.embedding_similarity
+ w["tfidf"] * self.tfidf_similarity
+ w["collaborative"] * self.collaborative_similarity
+ w["popularity"] * self.popularity_norm
)
return float(np.clip(score, 0.0, 1.0))
def rank_candidates(
candidates: List[RecommendationCandidate],
top_k: int = 5,
weights: Optional[Dict[str, float]] = None,
) -> List[Dict]:
scored = [(c, c.hybrid_score(weights)) for c in candidates]
scored.sort(key=lambda t: t[1], reverse=True)
return [
{
"brand": c.brand,
"image_id": c.image_id,
"similarity_score": round(score, 4),
"signals": {
"embedding_similarity": round(c.embedding_similarity, 4),
"tfidf_similarity": round(c.tfidf_similarity, 4),
"collaborative_similarity": round(c.collaborative_similarity, 4),
"popularity_norm": round(c.popularity_norm, 4),
},
}
for c, score in scored[:top_k]
]

View File

@@ -0,0 +1,126 @@
"""
Feature 9: Store Performance Prediction (Regression).
Predicts a store's expected revenue for the NEXT 7-day period from its
own trailing performance features (rolling revenue, order volume,
average basket value, discount depth, footfall proxy, tier, weekday
mix). Trained on real (not synthetic) simulated daily store aggregates
- like `forecasting.py`, this is genuine time-series-derived supervised
learning: the label is the store's actual future revenue in the
simulation, not a bootstrapped formula.
RandomForestRegressor: robust to the small number of stores (5) and
the resulting modest sample size once rolled up daily, without needing
heavy tuning.
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from app.intelligence import features as F
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
MODEL_NAME = "store_performance_model"
FEATURE_COLUMNS = [
"rolling_revenue_7", "rolling_revenue_14", "rolling_orders_7",
"avg_basket_value_7", "avg_discount_pct_7", "store_tier_encoded",
"month_sin", "month_cos",
]
def build_daily_store_aggregates(orders: pd.DataFrame, store_meta: pd.DataFrame) -> pd.DataFrame:
"""`orders` columns: store_id, order_date, order_value.
`store_meta` columns: store_id, tier.
Returns one row per (store_id, date) with revenue/order_count."""
if orders.empty:
return pd.DataFrame(columns=["store_id", "date", "revenue", "order_count"])
daily = orders.groupby(["store_id", orders["order_date"].dt.date]).agg(
revenue=("order_value", "sum"), order_count=("order_id", "count"),
).reset_index().rename(columns={"order_date": "date"})
daily["date"] = pd.to_datetime(daily["date"])
return daily.merge(store_meta, on="store_id", how="left")
def build_training_frame(daily_store_agg: pd.DataFrame, discount_avg_by_store_date: pd.DataFrame) -> pd.DataFrame:
"""`discount_avg_by_store_date` columns: store_id, date, avg_discount_pct."""
rows = []
merged = daily_store_agg.merge(discount_avg_by_store_date, on=["store_id", "date"], how="left")
merged["avg_discount_pct"] = merged["avg_discount_pct"].fillna(0.0)
for store_id, g in merged.sort_values("date").groupby("store_id"):
g = g.set_index("date")
revenue = g["revenue"].asfreq("D", fill_value=0)
orders_ct = g["order_count"].asfreq("D", fill_value=0)
discount = g["avg_discount_pct"].asfreq("D", fill_value=0)
tier_enc = F.encode_store_tier(g["tier"].iloc[0] if len(g) else "standard")
for i in range(21, len(revenue) - 7):
as_of = revenue.index[i]
future_revenue = revenue.iloc[i:i + 7].sum()
hist_rev = revenue.iloc[:i]
hist_orders = orders_ct.iloc[:i]
hist_disc = discount.iloc[:i]
basket = (hist_rev.tail(7).sum() / max(hist_orders.tail(7).sum(), 1))
rows.append({
"rolling_revenue_7": hist_rev.tail(7).mean(),
"rolling_revenue_14": hist_rev.tail(14).mean(),
"rolling_orders_7": hist_orders.tail(7).mean(),
"avg_basket_value_7": basket,
"avg_discount_pct_7": hist_disc.tail(7).mean(),
"store_tier_encoded": tier_enc,
**F.cyclical_month_features(as_of.date()),
"target_next7_revenue": future_revenue,
})
return pd.DataFrame(rows)
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.ensemble import RandomForestRegressor
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_absolute_error
X = training_frame[FEATURE_COLUMNS]
y = training_frame["target_next7_revenue"]
if len(training_frame) < 10:
# Too few samples (very short simulated history) for a train/test
# split to be meaningful - fit on everything and report NaN MAE
# rather than crashing.
model = RandomForestRegressor(n_estimators=100, max_depth=5, random_state=42, n_jobs=-1)
model.fit(X, y)
mae = float("nan")
else:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
model = RandomForestRegressor(n_estimators=200, max_depth=7, random_state=42, n_jobs=-1)
model.fit(X_train, y_train)
mae = float(mean_absolute_error(y_test, model.predict(X_test)))
bundle = ModelBundle(
estimator=model, feature_columns=FEATURE_COLUMNS, model_name=MODEL_NAME,
n_samples=len(training_frame), extra={"val_mae_revenue": round(mae, 2) if mae == mae else None},
)
save_bundle(bundle)
return bundle
class StorePerformancePredictor:
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def predict(self, feature_df: pd.DataFrame) -> pd.Series:
if not self._ensure_loaded() or feature_df.empty:
return pd.Series([0.0] * len(feature_df), index=feature_df.index)
X = feature_df[self._bundle.feature_columns]
preds = np.clip(self._bundle.estimator.predict(X), 0, None)
return pd.Series(preds, index=feature_df.index)
store_performance_predictor = StorePerformancePredictor()

View File

@@ -0,0 +1,217 @@
"""
Multi-store product distribution, pricing, and inventory generation.
Implements Feature 1 (Multi-Store Product Distribution + Store Pricing)
and Feature 2 (Product Stock Management) as pure functions over the
existing catalog data (read from pgvector via `services/store_db.py`,
never mutated here) - this module never touches the database itself, so
it's fully unit-testable and reusable from both the one-off seed script
and, if wanted later, an admin "reshuffle stores" endpoint.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Dict, List, Optional
import numpy as np
from app.intelligence.order_simulation import StoreProfile, deterministic_unit, latent_popularity
from app.services.price_estimator import classify_category, estimate_price, parse_price_string
# ---------------------------------------------------------------------------
# The 5 required stores. footfall_index is the baseline avg orders/day used
# by the order simulator (before weekday/seasonal multipliers) - premium
# stores in this simulated chain are smaller/boutique (lower footfall, higher
# price realization); standard/budget stores are higher-volume.
# ---------------------------------------------------------------------------
DEFAULT_STORES: List[StoreProfile] = [
StoreProfile("STORE-A", "Store-A - Gandhipuram", "Coimbatore", "premium", footfall_index=18),
StoreProfile("STORE-B", "Store-B - RS Puram", "Coimbatore", "standard", footfall_index=26),
StoreProfile("STORE-C", "Store-C - Peelamedu", "Coimbatore", "budget", footfall_index=34),
StoreProfile("STORE-D", "Store-D - Podanur", "Coimbatore", "standard", footfall_index=24),
StoreProfile("STORE-E", "Store-E - Ukkadam", "Coimbatore", "budget", footfall_index=30),
]
# Fraction of the full catalog each store stocks (Feature 1: "every
# product should not necessarily exist in every store"). Premium/boutique
# stores curate a smaller assortment; high-volume stores stock more.
STORE_ASSORTMENT_RATIO = {
"STORE-A": 0.55,
"STORE-B": 0.75,
"STORE-C": 0.85,
"STORE-D": 0.70,
"STORE-E": 0.80,
}
# Approximate typical Indian FMCG gross-margin bands per pricing category
# (from price_estimator.CATEGORY_BANDS keys). These are deliberately
# approximate, documented assumptions, not sourced from any single
# retailer's real books - used only to keep simulated cost prices
# realistic relative to MRP.
CATEGORY_COST_RATIO = {
"biscuits_cookies": 0.80, "crackers": 0.80, "rusk": 0.82, "bakery_bread": 0.78,
"cakes_muffins": 0.75, "chocolates": 0.78, "snacks_namkeen": 0.78, "dairy": 0.85,
"beverages_juice": 0.80, "beverages_tea_coffee": 0.76, "breakfast_cereal": 0.78,
"oral_care": 0.72, "hair_care": 0.70, "skin_bath": 0.72, "household_clean": 0.75,
"baby_care": 0.74, "general": 0.78,
}
# Store-tier pricing stance: premium stores price closer to MRP (less
# aggressive discounting on the shelf price itself - actual promotional
# discounts are handled separately by the ML discount model); budget/
# high-footfall stores price more competitively below MRP.
STORE_TIER_PRICE_FACTOR = {"premium": (0.97, 1.0), "standard": (0.93, 0.99), "budget": (0.88, 0.97)}
@dataclass
class ProductRef:
brand: str
image_id: str
title: str
category: Optional[str]
price_range: Optional[str]
@dataclass
class ProvisionedProduct:
store_id: str
brand: str
image_id: str
category: str
mrp: float
cost_price: float
selling_price: float
available_stock: int
reserved_stock: int
reorder_level: int
safety_stock: int
@property
def profit_margin(self) -> float:
return round(self.selling_price - self.cost_price, 2)
@property
def gross_profit_pct(self) -> float:
if self.selling_price <= 0:
return 0.0
return round((self.profit_margin / self.selling_price) * 100, 2)
@property
def markup_pct(self) -> float:
if self.cost_price <= 0:
return 0.0
return round((self.profit_margin / self.cost_price) * 100, 2)
@property
def stock_status(self) -> str:
if self.available_stock <= 0:
return "Out of Stock"
if self.available_stock <= self.safety_stock:
return "Low Stock"
if self.available_stock > self.reorder_level * 6:
return "Overstocked"
return "In Stock"
def resolve_mrp(product: ProductRef) -> float:
"""MRP is fixed per product (Feature 1 pricing rule: 'Keep MRP
fixed') - it never varies by store. Prefer the real price already
stored in the catalog (`price_range`, produced by the existing
price_estimator-backed ingestion pipeline); fall back to
`estimate_price` for a product with no usable price_range."""
parsed = parse_price_string(product.price_range or "")
if parsed and parsed > 0:
return float(parsed)
return float(estimate_price("100g", product.title, product.brand, product.category or ""))
def _stock_base_units(category_key: str, tier: str) -> int:
"""A category-appropriate baseline stock level before popularity and
store-tier scaling. Fast-moving low-unit-price categories (biscuits,
snacks) are stocked deeper than slow-moving/expensive categories."""
high_velocity = {"biscuits_cookies", "snacks_namkeen", "beverages_tea_coffee", "dairy", "crackers"}
base = 260 if category_key in high_velocity else 140
tier_mult = {"premium": 0.75, "standard": 1.0, "budget": 1.15}.get(tier, 1.0)
return int(base * tier_mult)
def provision_stores(
products: List[ProductRef],
stores: Optional[List[StoreProfile]] = None,
seed: int = 42,
) -> Dict[str, List[ProvisionedProduct]]:
"""Assign a random-but-reproducible subset of `products` to each
store, and generate independent per-store pricing + inventory for
every assigned product.
Returns {store_id: [ProvisionedProduct, ...]}.
"""
stores = stores or DEFAULT_STORES
rng = np.random.default_rng(seed)
result: Dict[str, List[ProvisionedProduct]] = {s.store_id: [] for s in stores}
for store in stores:
ratio = STORE_ASSORTMENT_RATIO.get(store.store_id, 0.70)
lo_factor, hi_factor = STORE_TIER_PRICE_FACTOR.get(store.tier, (0.92, 0.99))
for product in products:
# Deterministic-but-store-specific inclusion draw so re-running
# the seed script reproduces the same assortment.
pop = latent_popularity(product.brand, product.image_id)
inclusion_prob = min(0.98, ratio * (0.6 + 0.8 * pop)) # popular items more likely to be stocked everywhere
draw = deterministic_unit(f"assort|{store.store_id}|{product.brand}|{product.image_id}")
if draw > inclusion_prob:
continue
category_key = classify_category(product.title, product.category or "")
mrp = resolve_mrp(product)
cost_ratio = CATEGORY_COST_RATIO.get(category_key, 0.78)
cost_price = round(mrp * cost_ratio, 2)
# Per-store, per-product price jitter within the tier's factor
# band, seeded so it's stable across re-runs but differs across
# stores/products (Feature 1: "Price should vary between
# stores... Store prices independently").
jitter = deterministic_unit(f"price|{store.store_id}|{product.brand}|{product.image_id}")
factor = lo_factor + jitter * (hi_factor - lo_factor)
selling_price = round(mrp * factor, 2)
# Guardrail: never below a minimal viable margin, never above MRP.
min_viable = round(cost_price * 1.03, 2)
selling_price = max(min_viable, min(selling_price, mrp))
base_units = _stock_base_units(category_key, store.tier)
stock_jitter = rng.lognormal(mean=0.0, sigma=0.32)
available_stock = max(0, int(base_units * (0.55 + 0.7 * pop) * stock_jitter))
# ~4% of stores/products simulate a stockout so the "Out of
# Stock" status and its downstream discount/analytics effects
# actually show up in the demo data.
if deterministic_unit(f"oos|{store.store_id}|{product.brand}|{product.image_id}") < 0.04:
available_stock = 0
reorder_level = max(5, int(base_units * 0.30))
safety_stock = max(2, int(reorder_level * 0.5))
# ~10% of stocked (non-zero) rows simulate a "running low, not
# yet reordered" state so the Low Stock status and its
# downstream reorder-alert / higher-discount effects actually
# show up in the demo data instead of only In Stock/Overstocked.
if available_stock > 0 and deterministic_unit(
f"lowstock|{store.store_id}|{product.brand}|{product.image_id}"
) < 0.10:
available_stock = max(1, int(safety_stock * rng.uniform(0.3, 0.95)))
reserved_stock = int(available_stock * rng.uniform(0.0, 0.08))
result[store.store_id].append(ProvisionedProduct(
store_id=store.store_id,
brand=product.brand,
image_id=product.image_id,
category=product.category or "Uncategorized",
mrp=mrp,
cost_price=cost_price,
selling_price=selling_price,
available_stock=available_stock,
reserved_stock=reserved_stock,
reorder_level=reorder_level,
safety_stock=safety_stock,
))
return result

View File

@@ -0,0 +1,129 @@
"""
Synthetic training-label generators.
WHY THIS FILE EXISTS
---------------------
Two of the requested models - discount prediction and popularity scoring -
ask for a *regression* that predicts a number no historical dataset
actually contains yet ("what discount % SHOULD this product have had?",
"what SHOULD this product's popularity score be?"). There is no ground
truth for that in a brand-new system with no real transaction history.
The standard way to bootstrap a supervised model in this situation is:
1. Write down a transparent, multi-factor formula that encodes business
intuition (the one below mirrors the stock-tier example in the spec,
extended with demand/seasonality/expiry/category factors).
2. Add realistic noise to it, so the model doesn't just re-derive the
exact formula (which would make the "model" pointless) but instead
learns the *general relationship* between features and outcome,
including interactions the flat formula doesn't capture.
3. Train a regressor on the (features -> noisy-formula-label) pairs.
4. At INFERENCE time, only the trained model is used - never this
formula. New stock/demand/seasonal combinations the formula was
never explicitly tuned for still get a sensible prediction because
the model has generalized, and because the model also blends in the
real simulated sales-velocity/popularity features (which the flat
formula doesn't use fully), its predictions diverge from the raw
formula in exactly the way a supervised model is supposed to.
If/when this system accumulates real discount history (i.e. actual
markdowns and the resulting sales lift), `discount_model.py` should be
retrained on that real data instead and this module becomes unnecessary
for discounts. `popularity_model.py` can similarly be swapped to train on
real click/purchase telemetry once it exists.
"""
from __future__ import annotations
import numpy as np
import pandas as pd
def synthetic_discount_pct(df: pd.DataFrame, rng: np.random.Generator) -> pd.Series:
"""Formula-plus-noise target for discount %, in [0, 35].
Expected columns in `df`: stock_ratio, days_of_cover, units_per_day_30d,
demand_score (0-100), popularity_score (0-100), days_to_expiry,
is_festive_season (0/1), category_freq (0-1).
Directional logic (all learned relationships, not applied at
inference - see module docstring):
- More stock relative to reorder point -> higher discount (clear
the shelf).
- More days-of-cover than the category needs -> higher discount
(overstocked).
- Higher recent velocity / demand / popularity -> LOWER discount
(it's already selling, no need to discount it).
- Close to expiry -> higher discount (perishables urgency).
- Festive season -> a modest promotional discount bump.
- Niche/low-frequency categories get slightly higher clearance
discounts than high-turnover staples.
"""
stock_component = np.clip(df["stock_ratio"] * 9.0, 0, 22)
overstock_component = np.clip((df["days_of_cover"] - 20) * 0.35, 0, 10)
demand_relief = np.clip((df["demand_score"] + df["popularity_score"]) / 2.0 * 0.12, 0, 12)
expiry_component = np.where(
df["days_to_expiry"] <= 3, 14,
np.where(df["days_to_expiry"] <= 7, 8, np.where(df["days_to_expiry"] <= 14, 3, 0)),
)
festive_component = df["is_festive_season"] * 3.0
niche_component = (1.0 - df["category_freq"].clip(0, 1)) * 2.0
raw = (
stock_component
+ overstock_component
+ expiry_component
+ festive_component
+ niche_component
- demand_relief
)
noise = rng.normal(loc=0.0, scale=1.8, size=len(df))
return pd.Series(np.clip(raw + noise, 0, 35), index=df.index)
def synthetic_trend_score(df: pd.DataFrame, rng: np.random.Generator) -> pd.Series:
"""Formula-plus-noise target for a 0-100 'trending' score.
Expected columns: growth_pct (current vs previous window, can be
negative), revenue_growth_pct, order_count_current, recency_days
(days since the most recent order - lower is more 'alive right
now'), unique_customers_current.
A product is "trending" when it is both growing quickly AND has
enough absolute recent activity to be a meaningful signal (a jump
from 1 order to 2 orders is a 100% growth rate but not actually
trending) - the order_count/unique_customer terms exist so the
model learns to discount growth-rate spikes on near-zero volume,
which a naive "sort by growth %" rule (the literal hardcoded
approach the spec asks us to avoid) would get wrong.
"""
growth_component = np.clip(df["growth_pct"], -1, 5) * 12.0
revenue_component = np.clip(df["revenue_growth_pct"], -1, 5) * 8.0
volume_component = np.log1p(df["order_count_current"].clip(lower=0)) * 6.0
reach_component = np.log1p(df["unique_customers_current"].clip(lower=0)) * 5.0
recency_component = np.clip(14 - df["recency_days"], 0, 14) * 1.5
raw = growth_component + revenue_component + volume_component + reach_component + recency_component
noise = rng.normal(loc=0.0, scale=3.5, size=len(df))
scaled = np.clip(raw + noise, 0, None)
# Squash into 0-100 with a soft cap so a handful of extreme outliers
# don't compress everything else near zero.
return pd.Series(100 * (1 - np.exp(-scaled / 40.0)), index=df.index)
def synthetic_popularity_score(df: pd.DataFrame, rng: np.random.Generator) -> pd.Series:
"""Formula-plus-noise target for popularity, 0-100.
Expected columns: views_norm, wishlist_norm, orders_norm,
rating_norm, conversion_norm (all already 0-100 normalized).
Weighted blend chosen to reflect that actual purchases matter more
than passive views, mirroring typical e-commerce popularity scoring.
"""
raw = (
0.15 * df["views_norm"]
+ 0.15 * df["wishlist_norm"]
+ 0.40 * df["orders_norm"]
+ 0.15 * df["rating_norm"]
+ 0.15 * df["conversion_norm"]
)
noise = rng.normal(loc=0.0, scale=4.0, size=len(df))
return pd.Series(np.clip(raw + noise, 0, 100), index=df.index)

View File

@@ -0,0 +1,167 @@
"""
Feature 6: ML-Based Trending Product Detection.
Approach: rolling-window time-series feature engineering (current vs.
previous period unit/revenue growth, order frequency, customer reach,
recency) feeding a GradientBoostingRegressor that predicts a 0-100
trend score. This is one of the spec's own listed options ("Gradient
Boosting", "Random Forest") - chosen over Prophet/LSTM for the
hardware-conscious reasons in `app/intelligence/__init__.py`. The time
window aggregation (daily/weekly/monthly rollups, WoW/MoM growth) *is*
the time-series component; Prophet/LSTM would model the same rollups
with heavier machinery for a marginal accuracy gain that isn't worth
the RAM/CPU budget here.
Nothing is ever hardcoded as "the trending list" - every ranking below
is `predicted_score.sort_values(ascending=False)` on live order data.
"""
from __future__ import annotations
from datetime import date, timedelta
from typing import Dict, List, Literal, Optional
import numpy as np
import pandas as pd
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
from app.intelligence.synthetic_labels import synthetic_trend_score
MODEL_NAME = "trending_model"
FEATURE_COLUMNS = [
"growth_pct", "revenue_growth_pct", "order_count_current",
"unique_customers_current", "recency_days",
]
Window = Literal["today", "weekly", "monthly"]
WINDOW_DAYS: Dict[Window, int] = {"today": 1, "weekly": 7, "monthly": 30}
def _window_bounds(as_of: date, window: Window) -> tuple[pd.Timestamp, pd.Timestamp, pd.Timestamp]:
days = WINDOW_DAYS[window]
end = pd.Timestamp(as_of)
current_start = end - pd.Timedelta(days=days)
previous_start = current_start - pd.Timedelta(days=days)
return previous_start, current_start, end
def compute_trend_features(
order_items: pd.DataFrame,
orders: pd.DataFrame,
as_of: date,
window: Window,
group_cols: List[str],
) -> pd.DataFrame:
"""`group_cols` is either ['brand', 'image_id'] (overall/category
scope, pooled across stores) or ['store_id', 'brand', 'image_id']
(store-wise scope).
`order_items` must already carry `order_date` and `customer_id`
columns - this is the shape `store_db.get_order_items_df()` returns
(it joins those in from `orders` at the SQL layer so every caller
across this package can rely on the same enriched shape rather than
each re-joining separately). `orders` is accepted for API symmetry
with other builders in this module but isn't re-merged here.
Returns one row per group with FEATURE_COLUMNS populated from real
simulated order history - no synthetic data at this stage, only the
downstream label used for *training* is synthetic (see
synthetic_labels.py); features here are 100% derived from actual
simulated transactions.
"""
if order_items.empty or "order_date" not in order_items.columns:
return pd.DataFrame(columns=group_cols + FEATURE_COLUMNS)
merged = order_items
prev_start, cur_start, cur_end = _window_bounds(as_of, window)
current = merged[(merged["order_date"] >= cur_start) & (merged["order_date"] < cur_end)]
previous = merged[(merged["order_date"] >= prev_start) & (merged["order_date"] < cur_start)]
cur_agg = current.groupby(group_cols).agg(
units_current=("quantity", "sum"),
revenue_current=("line_total", "sum"),
order_count_current=("order_id", "nunique"),
unique_customers_current=("customer_id", "nunique"),
).reset_index()
prev_agg = previous.groupby(group_cols).agg(
units_previous=("quantity", "sum"),
revenue_previous=("line_total", "sum"),
).reset_index()
last_seen = merged.groupby(group_cols)["order_date"].max().reset_index().rename(columns={"order_date": "last_order_date"})
df = cur_agg.merge(prev_agg, on=group_cols, how="left").merge(last_seen, on=group_cols, how="left")
df[["units_previous", "revenue_previous"]] = df[["units_previous", "revenue_previous"]].fillna(0.0)
df["growth_pct"] = (df["units_current"] - df["units_previous"]) / df["units_previous"].replace(0, np.nan)
df["growth_pct"] = df["growth_pct"].fillna(df["units_current"].clip(upper=1.0)) # brand-new activity counts as modest growth, not undefined
df["revenue_growth_pct"] = (df["revenue_current"] - df["revenue_previous"]) / df["revenue_previous"].replace(0, np.nan)
df["revenue_growth_pct"] = df["revenue_growth_pct"].fillna(df["revenue_current"].clip(upper=1.0) / max(df["revenue_current"].max(), 1))
df["recency_days"] = (pd.Timestamp(as_of) - df["last_order_date"]).dt.days.clip(lower=0)
return df[group_cols + FEATURE_COLUMNS]
def train(training_frame: pd.DataFrame) -> ModelBundle:
from sklearn.ensemble import GradientBoostingRegressor
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_absolute_error
X = training_frame[FEATURE_COLUMNS]
y = training_frame["trend_score"]
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
model = GradientBoostingRegressor(n_estimators=150, max_depth=3, learning_rate=0.08, subsample=0.9, random_state=42)
model.fit(X_train, y_train)
mae = float(mean_absolute_error(y_test, model.predict(X_test)))
bundle = ModelBundle(
estimator=model, feature_columns=FEATURE_COLUMNS, model_name=MODEL_NAME,
n_samples=len(training_frame),
extra={"val_mae_points": round(mae, 3)},
)
save_bundle(bundle)
return bundle
def build_training_frame(order_items: pd.DataFrame, orders: pd.DataFrame, as_of_dates: List[date]) -> pd.DataFrame:
"""Builds training examples across several historical `as_of` cut
points and all three windows, so the model sees a range of
growth/recency patterns rather than a single snapshot."""
frames = []
for as_of in as_of_dates:
for window in ("today", "weekly", "monthly"):
f = compute_trend_features(order_items, orders, as_of, window, ["brand", "image_id"])
if not f.empty:
f["window"] = window
frames.append(f)
if not frames:
return pd.DataFrame(columns=FEATURE_COLUMNS + ["trend_score"])
df = pd.concat(frames, ignore_index=True)
rng = np.random.default_rng(11)
df["trend_score"] = synthetic_trend_score(df, rng)
return df
class TrendingScorer:
def __init__(self) -> None:
self._bundle: ModelBundle | None = None
def _ensure_loaded(self) -> bool:
if self._bundle is None:
self._bundle = load_bundle(MODEL_NAME)
return self._bundle is not None
def score(self, feature_df: pd.DataFrame) -> pd.Series:
"""Returns a 0-100 predicted trend score aligned to feature_df's
index. Falls back to a neutral 0.0 (never a hardcoded ranking)
if no model has been trained yet."""
if not self._ensure_loaded() or feature_df.empty:
return pd.Series([0.0] * len(feature_df), index=feature_df.index)
X = feature_df[self._bundle.feature_columns]
preds = np.clip(self._bundle.estimator.predict(X), 0, 100)
return pd.Series(preds, index=feature_df.index)
trending_scorer = TrendingScorer()