""" Shared, pure feature-engineering helpers. Nothing in this module touches the database or the network - every function takes plain Python / pandas structures in and returns them out, which is what lets `tests/test_intelligence.py` exercise the real ML feature logic without a live Postgres instance. """ from __future__ import annotations import math from datetime import date from typing import Dict, Iterable, Optional import pandas as pd # --------------------------------------------------------------------------- # Category -> shelf-life bucket (used for expiry-aware discounting). # Perishables get a short simulated shelf life; ambient/packaged goods get # a long one (effectively "doesn't expire for pricing purposes"). # --------------------------------------------------------------------------- PERISHABLE_SHELF_LIFE_DAYS = { "dairy": 10, "bakery & breads": 4, "cakes & muffins": 5, } DEFAULT_SHELF_LIFE_DAYS = 270 # ambient FMCG (biscuits, tea, soap, ...) STORE_TIER_ORDER = ["budget", "standard", "premium"] def shelf_life_days_for_category(category: Optional[str]) -> int: key = (category or "").strip().lower() for k, v in PERISHABLE_SHELF_LIFE_DAYS.items(): if k in key: return v return DEFAULT_SHELF_LIFE_DAYS def days_to_expiry(category: Optional[str], days_since_stocked: int) -> int: """Simulated remaining shelf life. Clamped at 0 (already expired stock would have been written off, so this floors at 0 rather than going negative).""" life = shelf_life_days_for_category(category) return max(0, life - int(days_since_stocked)) def cyclical_month_features(as_of: date) -> Dict[str, float]: """Sine/cosine encode the month so "December" and "January" are close in feature space (seasonality wraps around the year).""" angle = 2 * math.pi * (as_of.month - 1) / 12.0 return {"month_sin": math.sin(angle), "month_cos": math.cos(angle)} def cyclical_dow_features(as_of: date) -> Dict[str, float]: angle = 2 * math.pi * as_of.weekday() / 7.0 return {"dow_sin": math.sin(angle), "dow_cos": math.cos(angle)} def is_festive_season(as_of: date) -> int: """Coarse Indian FMCG festive-demand window (Oct-Nov: Diwali season; Aug: Independence Day/Onam-ish promo season). Used as a simple seasonal-trend signal rather than hardcoding a discount bump - the model learns how much this feature matters from the training data.""" return 1 if as_of.month in (10, 11) or as_of.month == 8 else 0 def stock_ratio(available_stock: int, reorder_level: int) -> float: """>1 means well-stocked relative to reorder point; <1 means at/below the reorder point. Reorder level is floored at 1 to avoid div-by-zero for misconfigured rows.""" return float(available_stock) / float(max(reorder_level, 1)) def days_of_cover(available_stock: int, avg_daily_sales: float) -> float: """How many days current stock would last at the recent sales pace. A very small floor on avg_daily_sales avoids an artificial 'infinite' days-of-cover for a product that just hasn't sold yet.""" return float(available_stock) / max(float(avg_daily_sales), 0.05) def sales_velocity(order_items: pd.DataFrame, store_id: str, brand: str, image_id: str, as_of: date, window_days: int) -> Dict[str, float]: """Units/day and revenue/day sold for one (store, product) over the trailing `window_days` ending at `as_of` (exclusive of future data - callers must only pass order history up to `as_of`). `order_items` is expected to have columns: store_id, brand, image_id, order_date (datetime64), quantity, line_total """ if order_items.empty: return {"units_per_day": 0.0, "revenue_per_day": 0.0, "order_count": 0} window_start = pd.Timestamp(as_of) - pd.Timedelta(days=window_days) mask = ( (order_items["store_id"] == store_id) & (order_items["brand"] == brand) & (order_items["image_id"] == image_id) & (order_items["order_date"] >= window_start) & (order_items["order_date"] < pd.Timestamp(as_of)) ) sub = order_items.loc[mask] units = float(sub["quantity"].sum()) revenue = float(sub["line_total"].sum()) return { "units_per_day": units / max(window_days, 1), "revenue_per_day": revenue / max(window_days, 1), "order_count": int(len(sub)), } def rfm_features(orders: pd.DataFrame, customer_id: str, as_of: date) -> Dict[str, float]: """Recency / Frequency / Monetary features for one customer, computed only from orders strictly before `as_of` (so this is safe to use as a training feature with a held-out future window as the label). `orders` columns: customer_id, order_date (datetime64), order_value """ hist = orders[(orders["customer_id"] == customer_id) & (orders["order_date"] < pd.Timestamp(as_of))] if hist.empty: return {"recency_days": 999.0, "frequency": 0.0, "monetary": 0.0, "avg_order_value": 0.0} last_order = hist["order_date"].max() recency_days = (pd.Timestamp(as_of) - last_order).days frequency = float(len(hist)) monetary = float(hist["order_value"].sum()) return { "recency_days": float(recency_days), "frequency": frequency, "monetary": monetary, "avg_order_value": monetary / frequency, } def normalize_0_100(series: pd.Series) -> pd.Series: """Min-max normalize a numeric series to a 0-100 range. Constant series map to 50 (avoids div-by-zero and avoids an arbitrary 0).""" lo, hi = series.min(), series.max() if hi - lo < 1e-9: return pd.Series([50.0] * len(series), index=series.index) return (series - lo) / (hi - lo) * 100.0 def encode_category(categories: Iterable[str]) -> pd.Series: """Simple stable frequency-encoding for a category column - keeps every model's category feature deterministic across train/inference without needing to persist a fitted OneHotEncoder for a small cardinality field.""" s = pd.Series(list(categories)).fillna("Uncategorized") freq = s.value_counts(normalize=True) return s.map(freq).fillna(0.0) def encode_store_tier(tier: str) -> int: try: return STORE_TIER_ORDER.index((tier or "standard").lower()) except ValueError: return 1