158 lines
6.2 KiB
Python
158 lines
6.2 KiB
Python
"""
|
|
Shared, pure feature-engineering helpers.
|
|
|
|
Nothing in this module touches the database or the network - every
|
|
function takes plain Python / pandas structures in and returns them
|
|
out, which is what lets `tests/test_intelligence.py` exercise the real
|
|
ML feature logic without a live Postgres instance.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import math
|
|
from datetime import date
|
|
from typing import Dict, Iterable, Optional
|
|
|
|
import pandas as pd
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Category -> shelf-life bucket (used for expiry-aware discounting).
|
|
# Perishables get a short simulated shelf life; ambient/packaged goods get
|
|
# a long one (effectively "doesn't expire for pricing purposes").
|
|
# ---------------------------------------------------------------------------
|
|
PERISHABLE_SHELF_LIFE_DAYS = {
|
|
"dairy": 10,
|
|
"bakery & breads": 4,
|
|
"cakes & muffins": 5,
|
|
}
|
|
DEFAULT_SHELF_LIFE_DAYS = 270 # ambient FMCG (biscuits, tea, soap, ...)
|
|
|
|
STORE_TIER_ORDER = ["budget", "standard", "premium"]
|
|
|
|
|
|
def shelf_life_days_for_category(category: Optional[str]) -> int:
|
|
key = (category or "").strip().lower()
|
|
for k, v in PERISHABLE_SHELF_LIFE_DAYS.items():
|
|
if k in key:
|
|
return v
|
|
return DEFAULT_SHELF_LIFE_DAYS
|
|
|
|
|
|
def days_to_expiry(category: Optional[str], days_since_stocked: int) -> int:
|
|
"""Simulated remaining shelf life. Clamped at 0 (already expired stock
|
|
would have been written off, so this floors at 0 rather than going
|
|
negative)."""
|
|
life = shelf_life_days_for_category(category)
|
|
return max(0, life - int(days_since_stocked))
|
|
|
|
|
|
def cyclical_month_features(as_of: date) -> Dict[str, float]:
|
|
"""Sine/cosine encode the month so "December" and "January" are close
|
|
in feature space (seasonality wraps around the year)."""
|
|
angle = 2 * math.pi * (as_of.month - 1) / 12.0
|
|
return {"month_sin": math.sin(angle), "month_cos": math.cos(angle)}
|
|
|
|
|
|
def cyclical_dow_features(as_of: date) -> Dict[str, float]:
|
|
angle = 2 * math.pi * as_of.weekday() / 7.0
|
|
return {"dow_sin": math.sin(angle), "dow_cos": math.cos(angle)}
|
|
|
|
|
|
def is_festive_season(as_of: date) -> int:
|
|
"""Coarse Indian FMCG festive-demand window (Oct-Nov: Diwali season;
|
|
Aug: Independence Day/Onam-ish promo season). Used as a simple
|
|
seasonal-trend signal rather than hardcoding a discount bump - the
|
|
model learns how much this feature matters from the training data."""
|
|
return 1 if as_of.month in (10, 11) or as_of.month == 8 else 0
|
|
|
|
|
|
def stock_ratio(available_stock: int, reorder_level: int) -> float:
|
|
""">1 means well-stocked relative to reorder point; <1 means at/below
|
|
the reorder point. Reorder level is floored at 1 to avoid div-by-zero
|
|
for misconfigured rows."""
|
|
return float(available_stock) / float(max(reorder_level, 1))
|
|
|
|
|
|
def days_of_cover(available_stock: int, avg_daily_sales: float) -> float:
|
|
"""How many days current stock would last at the recent sales pace.
|
|
A very small floor on avg_daily_sales avoids an artificial 'infinite'
|
|
days-of-cover for a product that just hasn't sold yet."""
|
|
return float(available_stock) / max(float(avg_daily_sales), 0.05)
|
|
|
|
|
|
def sales_velocity(order_items: pd.DataFrame, store_id: str, brand: str, image_id: str,
|
|
as_of: date, window_days: int) -> Dict[str, float]:
|
|
"""Units/day and revenue/day sold for one (store, product) over the
|
|
trailing `window_days` ending at `as_of` (exclusive of future data -
|
|
callers must only pass order history up to `as_of`).
|
|
|
|
`order_items` is expected to have columns:
|
|
store_id, brand, image_id, order_date (datetime64), quantity, line_total
|
|
"""
|
|
if order_items.empty:
|
|
return {"units_per_day": 0.0, "revenue_per_day": 0.0, "order_count": 0}
|
|
|
|
window_start = pd.Timestamp(as_of) - pd.Timedelta(days=window_days)
|
|
mask = (
|
|
(order_items["store_id"] == store_id)
|
|
& (order_items["brand"] == brand)
|
|
& (order_items["image_id"] == image_id)
|
|
& (order_items["order_date"] >= window_start)
|
|
& (order_items["order_date"] < pd.Timestamp(as_of))
|
|
)
|
|
sub = order_items.loc[mask]
|
|
units = float(sub["quantity"].sum())
|
|
revenue = float(sub["line_total"].sum())
|
|
return {
|
|
"units_per_day": units / max(window_days, 1),
|
|
"revenue_per_day": revenue / max(window_days, 1),
|
|
"order_count": int(len(sub)),
|
|
}
|
|
|
|
|
|
def rfm_features(orders: pd.DataFrame, customer_id: str, as_of: date) -> Dict[str, float]:
|
|
"""Recency / Frequency / Monetary features for one customer, computed
|
|
only from orders strictly before `as_of` (so this is safe to use as a
|
|
training feature with a held-out future window as the label).
|
|
|
|
`orders` columns: customer_id, order_date (datetime64), order_value
|
|
"""
|
|
hist = orders[(orders["customer_id"] == customer_id) & (orders["order_date"] < pd.Timestamp(as_of))]
|
|
if hist.empty:
|
|
return {"recency_days": 999.0, "frequency": 0.0, "monetary": 0.0, "avg_order_value": 0.0}
|
|
last_order = hist["order_date"].max()
|
|
recency_days = (pd.Timestamp(as_of) - last_order).days
|
|
frequency = float(len(hist))
|
|
monetary = float(hist["order_value"].sum())
|
|
return {
|
|
"recency_days": float(recency_days),
|
|
"frequency": frequency,
|
|
"monetary": monetary,
|
|
"avg_order_value": monetary / frequency,
|
|
}
|
|
|
|
|
|
def normalize_0_100(series: pd.Series) -> pd.Series:
|
|
"""Min-max normalize a numeric series to a 0-100 range. Constant
|
|
series map to 50 (avoids div-by-zero and avoids an arbitrary 0)."""
|
|
lo, hi = series.min(), series.max()
|
|
if hi - lo < 1e-9:
|
|
return pd.Series([50.0] * len(series), index=series.index)
|
|
return (series - lo) / (hi - lo) * 100.0
|
|
|
|
|
|
def encode_category(categories: Iterable[str]) -> pd.Series:
|
|
"""Simple stable frequency-encoding for a category column - keeps
|
|
every model's category feature deterministic across train/inference
|
|
without needing to persist a fitted OneHotEncoder for a small
|
|
cardinality field."""
|
|
s = pd.Series(list(categories)).fillna("Uncategorized")
|
|
freq = s.value_counts(normalize=True)
|
|
return s.map(freq).fillna(0.0)
|
|
|
|
|
|
def encode_store_tier(tier: str) -> int:
|
|
try:
|
|
return STORE_TIER_ORDER.index((tier or "standard").lower())
|
|
except ValueError:
|
|
return 1
|