updates on the backend
This commit is contained in:
158
app/intelligence/features.py
Normal file
158
app/intelligence/features.py
Normal file
@@ -0,0 +1,158 @@
|
||||
"""
|
||||
Shared, pure feature-engineering helpers.
|
||||
|
||||
Nothing in this module touches the database or the network - every
|
||||
function takes plain Python / pandas structures in and returns them
|
||||
out, which is what lets `tests/test_intelligence.py` exercise the real
|
||||
ML feature logic without a live Postgres instance.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from datetime import date, datetime
|
||||
from typing import Dict, Iterable, List, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Category -> shelf-life bucket (used for expiry-aware discounting).
|
||||
# Perishables get a short simulated shelf life; ambient/packaged goods get
|
||||
# a long one (effectively "doesn't expire for pricing purposes").
|
||||
# ---------------------------------------------------------------------------
|
||||
PERISHABLE_SHELF_LIFE_DAYS = {
|
||||
"dairy": 10,
|
||||
"bakery & breads": 4,
|
||||
"cakes & muffins": 5,
|
||||
}
|
||||
DEFAULT_SHELF_LIFE_DAYS = 270 # ambient FMCG (biscuits, tea, soap, ...)
|
||||
|
||||
STORE_TIER_ORDER = ["budget", "standard", "premium"]
|
||||
|
||||
|
||||
def shelf_life_days_for_category(category: Optional[str]) -> int:
|
||||
key = (category or "").strip().lower()
|
||||
for k, v in PERISHABLE_SHELF_LIFE_DAYS.items():
|
||||
if k in key:
|
||||
return v
|
||||
return DEFAULT_SHELF_LIFE_DAYS
|
||||
|
||||
|
||||
def days_to_expiry(category: Optional[str], days_since_stocked: int) -> int:
|
||||
"""Simulated remaining shelf life. Clamped at 0 (already expired stock
|
||||
would have been written off, so this floors at 0 rather than going
|
||||
negative)."""
|
||||
life = shelf_life_days_for_category(category)
|
||||
return max(0, life - int(days_since_stocked))
|
||||
|
||||
|
||||
def cyclical_month_features(as_of: date) -> Dict[str, float]:
|
||||
"""Sine/cosine encode the month so "December" and "January" are close
|
||||
in feature space (seasonality wraps around the year)."""
|
||||
angle = 2 * math.pi * (as_of.month - 1) / 12.0
|
||||
return {"month_sin": math.sin(angle), "month_cos": math.cos(angle)}
|
||||
|
||||
|
||||
def cyclical_dow_features(as_of: date) -> Dict[str, float]:
|
||||
angle = 2 * math.pi * as_of.weekday() / 7.0
|
||||
return {"dow_sin": math.sin(angle), "dow_cos": math.cos(angle)}
|
||||
|
||||
|
||||
def is_festive_season(as_of: date) -> int:
|
||||
"""Coarse Indian FMCG festive-demand window (Oct-Nov: Diwali season;
|
||||
Aug: Independence Day/Onam-ish promo season). Used as a simple
|
||||
seasonal-trend signal rather than hardcoding a discount bump - the
|
||||
model learns how much this feature matters from the training data."""
|
||||
return 1 if as_of.month in (10, 11) or as_of.month == 8 else 0
|
||||
|
||||
|
||||
def stock_ratio(available_stock: int, reorder_level: int) -> float:
|
||||
""">1 means well-stocked relative to reorder point; <1 means at/below
|
||||
the reorder point. Reorder level is floored at 1 to avoid div-by-zero
|
||||
for misconfigured rows."""
|
||||
return float(available_stock) / float(max(reorder_level, 1))
|
||||
|
||||
|
||||
def days_of_cover(available_stock: int, avg_daily_sales: float) -> float:
|
||||
"""How many days current stock would last at the recent sales pace.
|
||||
A very small floor on avg_daily_sales avoids an artificial 'infinite'
|
||||
days-of-cover for a product that just hasn't sold yet."""
|
||||
return float(available_stock) / max(float(avg_daily_sales), 0.05)
|
||||
|
||||
|
||||
def sales_velocity(order_items: pd.DataFrame, store_id: str, brand: str, image_id: str,
|
||||
as_of: date, window_days: int) -> Dict[str, float]:
|
||||
"""Units/day and revenue/day sold for one (store, product) over the
|
||||
trailing `window_days` ending at `as_of` (exclusive of future data -
|
||||
callers must only pass order history up to `as_of`).
|
||||
|
||||
`order_items` is expected to have columns:
|
||||
store_id, brand, image_id, order_date (datetime64), quantity, line_total
|
||||
"""
|
||||
if order_items.empty:
|
||||
return {"units_per_day": 0.0, "revenue_per_day": 0.0, "order_count": 0}
|
||||
|
||||
window_start = pd.Timestamp(as_of) - pd.Timedelta(days=window_days)
|
||||
mask = (
|
||||
(order_items["store_id"] == store_id)
|
||||
& (order_items["brand"] == brand)
|
||||
& (order_items["image_id"] == image_id)
|
||||
& (order_items["order_date"] >= window_start)
|
||||
& (order_items["order_date"] < pd.Timestamp(as_of))
|
||||
)
|
||||
sub = order_items.loc[mask]
|
||||
units = float(sub["quantity"].sum())
|
||||
revenue = float(sub["line_total"].sum())
|
||||
return {
|
||||
"units_per_day": units / max(window_days, 1),
|
||||
"revenue_per_day": revenue / max(window_days, 1),
|
||||
"order_count": int(len(sub)),
|
||||
}
|
||||
|
||||
|
||||
def rfm_features(orders: pd.DataFrame, customer_id: str, as_of: date) -> Dict[str, float]:
|
||||
"""Recency / Frequency / Monetary features for one customer, computed
|
||||
only from orders strictly before `as_of` (so this is safe to use as a
|
||||
training feature with a held-out future window as the label).
|
||||
|
||||
`orders` columns: customer_id, order_date (datetime64), order_value
|
||||
"""
|
||||
hist = orders[(orders["customer_id"] == customer_id) & (orders["order_date"] < pd.Timestamp(as_of))]
|
||||
if hist.empty:
|
||||
return {"recency_days": 999.0, "frequency": 0.0, "monetary": 0.0, "avg_order_value": 0.0}
|
||||
last_order = hist["order_date"].max()
|
||||
recency_days = (pd.Timestamp(as_of) - last_order).days
|
||||
frequency = float(len(hist))
|
||||
monetary = float(hist["order_value"].sum())
|
||||
return {
|
||||
"recency_days": float(recency_days),
|
||||
"frequency": frequency,
|
||||
"monetary": monetary,
|
||||
"avg_order_value": monetary / frequency,
|
||||
}
|
||||
|
||||
|
||||
def normalize_0_100(series: pd.Series) -> pd.Series:
|
||||
"""Min-max normalize a numeric series to a 0-100 range. Constant
|
||||
series map to 50 (avoids div-by-zero and avoids an arbitrary 0)."""
|
||||
lo, hi = series.min(), series.max()
|
||||
if hi - lo < 1e-9:
|
||||
return pd.Series([50.0] * len(series), index=series.index)
|
||||
return (series - lo) / (hi - lo) * 100.0
|
||||
|
||||
|
||||
def encode_category(categories: Iterable[str]) -> pd.Series:
|
||||
"""Simple stable frequency-encoding for a category column - keeps
|
||||
every model's category feature deterministic across train/inference
|
||||
without needing to persist a fitted OneHotEncoder for a small
|
||||
cardinality field."""
|
||||
s = pd.Series(list(categories)).fillna("Uncategorized")
|
||||
freq = s.value_counts(normalize=True)
|
||||
return s.map(freq).fillna(0.0)
|
||||
|
||||
|
||||
def encode_store_tier(tier: str) -> int:
|
||||
try:
|
||||
return STORE_TIER_ORDER.index((tier or "standard").lower())
|
||||
except ValueError:
|
||||
return 1
|
||||
Reference in New Issue
Block a user