Compare commits
2 Commits
f804791864
...
4d21b44ddc
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d21b44ddc | ||
|
|
9ef2f61870 |
@@ -38,6 +38,9 @@ DEFAULTS: Dict[str, Any] = {
|
|||||||
"emergency_load_penalty": 3.0, # km penalty per order in emergency assign
|
"emergency_load_penalty": 3.0, # km penalty per order in emergency assign
|
||||||
# RouteOptimizer
|
# RouteOptimizer
|
||||||
"search_time_limit_seconds": 5,
|
"search_time_limit_seconds": 5,
|
||||||
|
# Active-rider roster cache. The upstream getriderlogs call measured ~6s on
|
||||||
|
# live traffic. Short TTL so on/off-duty changes surface quickly; 0 = off.
|
||||||
|
"rider_roster_cache_ttl_seconds": 30,
|
||||||
"avg_speed_kmh": 18.0,
|
"avg_speed_kmh": 18.0,
|
||||||
"road_factor": 1.3,
|
"road_factor": 1.3,
|
||||||
# ClusteringService
|
# ClusteringService
|
||||||
@@ -56,12 +59,20 @@ DEFAULTS: Dict[str, Any] = {
|
|||||||
"eta_history_days": 14, # rolling window pulled from nearledb
|
"eta_history_days": 14, # rolling window pulled from nearledb
|
||||||
"eta_stat": "median", # "median" or "p75" (p75 = more conservative)
|
"eta_stat": "median", # "median" or "p75" (p75 = more conservative)
|
||||||
"eta_sync_interval_hours": 6, # autonomous background sync cadence
|
"eta_sync_interval_hours": 6, # autonomous background sync cadence
|
||||||
# Road-aware sequencing (Phase 2). OFF by default: enabling adds a Google
|
# Road-aware sequencing (Phase 2). Backed by self-hosted Valhalla, so there is
|
||||||
# Directions call (cost + latency) to the route hot path. Results are cached.
|
# no per-request cost — only local latency. Still OFF by default and left to
|
||||||
# Only the *visiting order* changes; step/ETA metrics stay aerial-based.
|
# the agent to enable on measured gain. Only the *visiting order* changes;
|
||||||
|
# step/ETA metrics stay aerial-based.
|
||||||
"routing_use_road_distance": False, # AGENT-MANAGED (see routing_auto_manage)
|
"routing_use_road_distance": False, # AGENT-MANAGED (see routing_auto_manage)
|
||||||
"routing_road_cache_ttl_seconds": 86400, # road geometry is stable; cache 24h
|
"routing_road_cache_ttl_seconds": 86400, # road geometry is stable; cache 24h
|
||||||
"routing_road_max_stops": 25, # Google distance-matrix practical cap
|
"routing_road_max_stops": 25, # beyond this the solver time outweighs
|
||||||
|
# the ordering gain (not a backend cap)
|
||||||
|
# Valhalla matrix backend. "motorcycle" models the lane access and one-way
|
||||||
|
# behaviour our riders actually have; "auto" would overstate their travel time.
|
||||||
|
"routing_valhalla_costing": "motorcycle",
|
||||||
|
"routing_matrix_timeout_seconds": 15.0,
|
||||||
|
"routing_matrix_max_unroutable_pct": 20.0, # above this -> tiles likely missing
|
||||||
|
# this region, fall back to aerial
|
||||||
# Autonomous road-sequencing decision agent: measures road-vs-aerial travel
|
# Autonomous road-sequencing decision agent: measures road-vs-aerial travel
|
||||||
# time on real batches and flips routing_use_road_distance on its own.
|
# time on real batches and flips routing_use_road_distance on its own.
|
||||||
"routing_auto_manage": True, # False -> humans own the flag
|
"routing_auto_manage": True, # False -> humans own the flag
|
||||||
|
|||||||
23
app/main.py
23
app/main.py
@@ -6,7 +6,7 @@ import sys
|
|||||||
import time
|
import time
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
|
|
||||||
# Load .env (DB_*, REDIS_*, GOOGLE_MAPS_API_KEY) into the environment early,
|
# Load .env (DB_*, REDIS_*, VALHALLA_URL) into the environment early,
|
||||||
# before any service reads os.getenv at import time.
|
# before any service reads os.getenv at import time.
|
||||||
try:
|
try:
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
@@ -111,6 +111,27 @@ async def lifespan(app: FastAPI):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning(f"[RoadAgent] Decision agent start failed (non-fatal): {e}")
|
logger.warning(f"[RoadAgent] Decision agent start failed (non-fatal): {e}")
|
||||||
|
|
||||||
|
# Probe the Valhalla matrix backend once. Road sequencing degrades to aerial
|
||||||
|
# silently by design, so without this an unreachable or still-building backend
|
||||||
|
# stays invisible — which is exactly how the expired Google key went unnoticed.
|
||||||
|
try:
|
||||||
|
from app.services.routing.route_optimizer import road_backend_status
|
||||||
|
_rb = await road_backend_status()
|
||||||
|
if not _rb.get("configured"):
|
||||||
|
logger.warning("[RoadMatrix] VALHALLA_URL not set - aerial sequencing only.")
|
||||||
|
elif _rb.get("reachable"):
|
||||||
|
logger.info(
|
||||||
|
f"[RoadMatrix] Valhalla ready at {_rb.get('url')} "
|
||||||
|
f"(version={_rb.get('version')}, tiles={_rb.get('tileset_last_modified')})"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"[RoadMatrix] Valhalla at {_rb.get('url')} unreachable "
|
||||||
|
f"({_rb.get('detail')}) - aerial fallback until it answers."
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"[RoadMatrix] Backend probe failed (non-fatal): {e}")
|
||||||
|
|
||||||
logger.info("[OK] Application ready.")
|
logger.info("[OK] Application ready.")
|
||||||
yield
|
yield
|
||||||
logger.info("[STOP] Route Optimization API shutting down.")
|
logger.info("[STOP] Route Optimization API shutting down.")
|
||||||
|
|||||||
@@ -84,6 +84,26 @@ async def readiness_check(request: Request):
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/road-backend")
|
||||||
|
async def road_backend_check(request: Request):
|
||||||
|
"""
|
||||||
|
Status of the Valhalla road-matrix backend used for road-aware stop sequencing.
|
||||||
|
|
||||||
|
Deliberately separate from /ready: an unreachable backend is a degradation,
|
||||||
|
not an outage — routing falls back to aerial ordering and keeps serving. This
|
||||||
|
endpoint is how you notice the degradation.
|
||||||
|
"""
|
||||||
|
from app.services.routing.route_optimizer import road_backend_status
|
||||||
|
|
||||||
|
status = await road_backend_status()
|
||||||
|
return {
|
||||||
|
**status,
|
||||||
|
"sequencing_mode": "road" if status.get("reachable") else "aerial",
|
||||||
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
||||||
|
"request_id": getattr(request.state, "request_id", None),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@router.get("/live")
|
@router.get("/live")
|
||||||
async def liveness_check(request: Request):
|
async def liveness_check(request: Request):
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -15,6 +15,31 @@ except Exception: # pragma: no cover
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _redis_url_from_parts() -> Optional[str]:
|
||||||
|
"""
|
||||||
|
Build a redis:// URL from discrete env vars, percent-encoding the credentials.
|
||||||
|
|
||||||
|
A hand-written REDIS_URL breaks the moment the password contains a reserved
|
||||||
|
character: an '@' splits the authority section early, so the client silently
|
||||||
|
dials a garbage host instead of failing loudly. Assembling the URL here with
|
||||||
|
quote() removes that whole class of bug - callers set REDIS_HOST/REDIS_PASSWORD
|
||||||
|
and never have to think about escaping.
|
||||||
|
|
||||||
|
Returns None when REDIS_HOST is unset, so an explicit REDIS_URL still wins.
|
||||||
|
"""
|
||||||
|
host = os.getenv("REDIS_HOST")
|
||||||
|
if not host:
|
||||||
|
return None
|
||||||
|
from urllib.parse import quote
|
||||||
|
|
||||||
|
user = os.getenv("REDIS_USERNAME", "")
|
||||||
|
pwd = os.getenv("REDIS_PASSWORD", "")
|
||||||
|
port = os.getenv("REDIS_PORT", "6379")
|
||||||
|
db = os.getenv("REDIS_DB", "0")
|
||||||
|
auth = f"{quote(user, safe='')}:{quote(pwd, safe='')}@" if pwd else ""
|
||||||
|
return f"redis://{auth}{host}:{port}/{db}"
|
||||||
|
|
||||||
|
|
||||||
class RedisCache:
|
class RedisCache:
|
||||||
"""Lightweight Redis cache wrapper with graceful in-memory fallback."""
|
"""Lightweight Redis cache wrapper with graceful in-memory fallback."""
|
||||||
|
|
||||||
@@ -33,9 +58,10 @@ class RedisCache:
|
|||||||
self._client = None
|
self._client = None
|
||||||
self._stats = {"hits": 0, "misses": 0, "sets": 0}
|
self._stats = {"hits": 0, "misses": 0, "sets": 0}
|
||||||
|
|
||||||
url = os.getenv(url_env)
|
url = os.getenv(url_env) or _redis_url_from_parts()
|
||||||
if not url or redis is None:
|
if not url or redis is None:
|
||||||
logger.warning("Redis not configured or client unavailable; falling back to local thread-safe in-memory cache")
|
_why = "redis client not installed" if redis is None else "no REDIS_URL/REDIS_HOST"
|
||||||
|
logger.warning(f"Redis not configured or client unavailable ({_why}); falling back to local thread-safe in-memory cache")
|
||||||
return
|
return
|
||||||
try:
|
try:
|
||||||
self._client = redis.Redis.from_url(url, decode_responses=True)
|
self._client = redis.Redis.from_url(url, decode_responses=True)
|
||||||
|
|||||||
@@ -6,13 +6,45 @@ from typing import List, Dict, Any
|
|||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
def _roster_cache_ttl() -> int:
|
||||||
|
"""Roster cache TTL in seconds. 0 disables caching entirely."""
|
||||||
|
try:
|
||||||
|
from app.config.dynamic_config import get_config
|
||||||
|
return int(get_config().get("rider_roster_cache_ttl_seconds", 30))
|
||||||
|
except Exception:
|
||||||
|
return 30
|
||||||
|
|
||||||
|
|
||||||
async def fetch_active_riders() -> List[Dict[str, Any]]:
|
async def fetch_active_riders() -> List[Dict[str, Any]]:
|
||||||
"""
|
"""
|
||||||
Fetch active rider logs from the external API for the current date.
|
Fetch active rider logs from the external API for the current date.
|
||||||
Returns a list of rider log dictionaries.
|
Returns a list of rider log dictionaries.
|
||||||
|
|
||||||
|
This upstream call measured ~6s on live traffic and was the single largest
|
||||||
|
component of riderassign latency, so successful responses are cached for
|
||||||
|
`rider_roster_cache_ttl_seconds`. The TTL is deliberately short: a rider
|
||||||
|
going on or off duty must become visible quickly, and the absent-rider
|
||||||
|
validation in the assignment path checks against this roster.
|
||||||
|
|
||||||
|
Only non-empty successes are cached. Caching an empty result or a failure
|
||||||
|
would turn a brief upstream blip into a full-TTL outage where every request
|
||||||
|
sees zero riders.
|
||||||
"""
|
"""
|
||||||
|
today_str = datetime.now().strftime("%Y-%m-%d")
|
||||||
|
ttl = _roster_cache_ttl()
|
||||||
|
cache_key = f"riders:active:{today_str}"
|
||||||
|
|
||||||
|
if ttl > 0:
|
||||||
|
try:
|
||||||
|
from app.services import cache as _cache
|
||||||
|
cached = _cache.get_json(cache_key)
|
||||||
|
if isinstance(cached, list) and cached:
|
||||||
|
logger.debug(f"[Riders] roster cache hit ({len(cached)} riders)")
|
||||||
|
return cached
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
try:
|
try:
|
||||||
today_str = datetime.now().strftime("%Y-%m-%d")
|
|
||||||
url = "https://jupiter.nearle.app/live/api/v2/partners/getriderlogs/"
|
url = "https://jupiter.nearle.app/live/api/v2/partners/getriderlogs/"
|
||||||
params = {
|
params = {
|
||||||
"applocationid": 1,
|
"applocationid": 1,
|
||||||
@@ -32,7 +64,14 @@ async def fetch_active_riders() -> List[Dict[str, Any]]:
|
|||||||
# The user's example showed "onduty": 1. We might want to filter by that.
|
# The user's example showed "onduty": 1. We might want to filter by that.
|
||||||
# For now, returning all logs, filtering can happen in assignment logic or here.
|
# For now, returning all logs, filtering can happen in assignment logic or here.
|
||||||
# Let's return the raw list as requested, filtering logic will be applied during assignment.
|
# Let's return the raw list as requested, filtering logic will be applied during assignment.
|
||||||
return data.get("details", [])
|
riders = data.get("details", [])
|
||||||
|
if riders and ttl > 0:
|
||||||
|
try:
|
||||||
|
from app.services import cache as _cache
|
||||||
|
_cache.set_json(cache_key, riders, ttl_seconds=ttl)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return riders
|
||||||
|
|
||||||
logger.warning(f"Fetch active riders returned no details: {data}")
|
logger.warning(f"Fetch active riders returned no details: {data}")
|
||||||
return []
|
return []
|
||||||
|
|||||||
@@ -61,8 +61,8 @@ class RoadSequencingAgent:
|
|||||||
from app.core.arrow_utils import calculate_haversine_matrix_vectorized
|
from app.core.arrow_utils import calculate_haversine_matrix_vectorized
|
||||||
|
|
||||||
opt = self._optimizer()
|
opt = self._optimizer()
|
||||||
if not opt.use_google_maps:
|
if not opt.use_road_matrix:
|
||||||
return {"evaluated": 0, "reason": "no_google_key", "mean_gain_pct": 0.0}
|
return {"evaluated": 0, "reason": "no_matrix_backend", "mean_gain_pct": 0.0}
|
||||||
|
|
||||||
batches = get_delivery_history_service().sample_batches(
|
batches = get_delivery_history_service().sample_batches(
|
||||||
days=days, limit=sample_batches
|
days=days, limit=sample_batches
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ ALGORITHM: TSP / VRP with Google OR-Tools
|
|||||||
|
|
||||||
FEATURES:
|
FEATURES:
|
||||||
- Automatic outlier detection and coordinate correction
|
- Automatic outlier detection and coordinate correction
|
||||||
- Hybrid distance calculation (Google Maps + Haversine fallback)
|
- Hybrid distance calculation (self-hosted Valhalla road matrix + Haversine fallback)
|
||||||
- Robust error handling for invalid inputs
|
- Robust error handling for invalid inputs
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -37,6 +37,37 @@ except ImportError:
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def road_backend_status() -> Dict[str, Any]:
|
||||||
|
"""
|
||||||
|
Probe the Valhalla matrix backend.
|
||||||
|
|
||||||
|
Module-level (not a RouteOptimizer method) so startup and health checks can
|
||||||
|
call it without paying for a full optimizer construction, which pulls in the
|
||||||
|
empirical ETA calculator and its history load.
|
||||||
|
"""
|
||||||
|
url = os.getenv("VALHALLA_URL", "").strip().rstrip("/")
|
||||||
|
if not url:
|
||||||
|
return {
|
||||||
|
"configured": False,
|
||||||
|
"reachable": False,
|
||||||
|
"detail": "VALHALLA_URL not set - road sequencing disabled, aerial only",
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||||
|
resp = await client.get(f"{url}/status")
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
return {
|
||||||
|
"configured": True,
|
||||||
|
"reachable": True,
|
||||||
|
"url": url,
|
||||||
|
"version": data.get("version"),
|
||||||
|
"tileset_last_modified": data.get("tileset_last_modified"),
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
return {"configured": True, "reachable": False, "url": url, "detail": str(e)}
|
||||||
|
|
||||||
|
|
||||||
class RouteOptimizer:
|
class RouteOptimizer:
|
||||||
"""Route optimization using Google OR-Tools (Async)."""
|
"""Route optimization using Google OR-Tools (Async)."""
|
||||||
|
|
||||||
@@ -60,9 +91,11 @@ class RouteOptimizer:
|
|||||||
# Road factor (haversine -> road distance multiplier, ML-tuned)
|
# Road factor (haversine -> road distance multiplier, ML-tuned)
|
||||||
self.road_factor = float(_cfg.get("road_factor"))
|
self.road_factor = float(_cfg.get("road_factor"))
|
||||||
|
|
||||||
# Google Maps API settings
|
# Road travel-time matrix backend: self-hosted Valhalla (no per-request
|
||||||
self.google_maps_api_key = os.getenv("GOOGLE_MAPS_API_KEY", "")
|
# cost, no rate limit). Unset VALHALLA_URL -> road sequencing stays off
|
||||||
self.use_google_maps = bool(self.google_maps_api_key)
|
# and every caller falls back to aerial ordering.
|
||||||
|
self.valhalla_url = os.getenv("VALHALLA_URL", "").strip().rstrip("/")
|
||||||
|
self.use_road_matrix = bool(self.valhalla_url)
|
||||||
|
|
||||||
# Solver time limit (ML-tuned)
|
# Solver time limit (ML-tuned)
|
||||||
self.search_time_limit_seconds = int(_cfg.get("search_time_limit_seconds"))
|
self.search_time_limit_seconds = int(_cfg.get("search_time_limit_seconds"))
|
||||||
@@ -90,47 +123,94 @@ class RouteOptimizer:
|
|||||||
# ROAD-AWARE VISITING ORDER (Phase 2 - opt-in, cached)
|
# ROAD-AWARE VISITING ORDER (Phase 2 - opt-in, cached)
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _aerial_minutes(
|
||||||
|
self, a: Tuple[float, float], b: Tuple[float, float]
|
||||||
|
) -> float:
|
||||||
|
"""Haversine travel-time estimate (minutes), used to patch unroutable pairs."""
|
||||||
|
km = self.haversine_distance(a[0], a[1], b[0], b[1]) * self.road_factor
|
||||||
|
return (km / (self.avg_speed_kmh or 20.0)) * 60.0
|
||||||
|
|
||||||
async def _road_duration_matrix(
|
async def _road_duration_matrix(
|
||||||
self, coords: _List[Tuple[float, float]]
|
self, coords: _List[Tuple[float, float]]
|
||||||
) -> Optional[_List[_List[float]]]:
|
) -> Optional[_List[_List[float]]]:
|
||||||
"""
|
"""
|
||||||
Full NxN road travel-TIME matrix (minutes) via Google Distance Matrix.
|
Full NxN road travel-TIME matrix (minutes) via self-hosted Valhalla.
|
||||||
|
|
||||||
|
One /sources_to_targets call returns the whole matrix, so unlike the old
|
||||||
|
Google Distance Matrix path there is no per-request element cap and no
|
||||||
|
chunking. Costing defaults to `motorcycle`, which models the lane access
|
||||||
|
and one-way behaviour our riders actually have — a car matrix systematically
|
||||||
|
overstates their travel time.
|
||||||
|
|
||||||
|
Valhalla reports an unroutable pair as time=null. Those are patched with an
|
||||||
|
aerial estimate rather than left at 0, because a 0-cost edge would look
|
||||||
|
free to the TSP and pull the whole sequence through it. If too many pairs
|
||||||
|
are unroutable the tileset probably doesn't cover this region, so we bail
|
||||||
|
to aerial entirely.
|
||||||
|
|
||||||
Chunks destinations to respect Google's ~100-elements-per-request limit.
|
|
||||||
Returns None on any failure so the caller falls back to aerial ordering.
|
Returns None on any failure so the caller falls back to aerial ordering.
|
||||||
"""
|
"""
|
||||||
if not self.use_google_maps:
|
if not self.use_road_matrix:
|
||||||
return None
|
return None
|
||||||
|
cfg = get_config()
|
||||||
|
costing = str(cfg.get("routing_valhalla_costing", "motorcycle"))
|
||||||
|
timeout = float(cfg.get("routing_matrix_timeout_seconds", 15.0))
|
||||||
|
max_unroutable = float(cfg.get("routing_matrix_max_unroutable_pct", 20.0))
|
||||||
|
|
||||||
n = len(coords)
|
n = len(coords)
|
||||||
origins = "|".join(f"{la},{lo}" for la, lo in coords)
|
locations = [{"lat": float(la), "lon": float(lo)} for la, lo in coords]
|
||||||
matrix = [[0.0] * n for _ in range(n)]
|
matrix = [[0.0] * n for _ in range(n)]
|
||||||
dest_chunk = max(1, 100 // max(1, n))
|
|
||||||
try:
|
try:
|
||||||
async with httpx.AsyncClient(timeout=15.0) as client:
|
async with httpx.AsyncClient(timeout=timeout) as client:
|
||||||
for j0 in range(0, n, dest_chunk):
|
resp = await client.post(
|
||||||
js = list(range(j0, min(j0 + dest_chunk, n)))
|
f"{self.valhalla_url}/sources_to_targets",
|
||||||
dests = "|".join(f"{coords[j][0]},{coords[j][1]}" for j in js)
|
json={
|
||||||
resp = await client.get(
|
"sources": locations,
|
||||||
"https://maps.googleapis.com/maps/api/distancematrix/json",
|
"targets": locations,
|
||||||
params={"origins": origins, "destinations": dests,
|
"costing": costing,
|
||||||
"key": self.google_maps_api_key, "units": "metric"},
|
"units": "km",
|
||||||
)
|
},
|
||||||
resp.raise_for_status()
|
)
|
||||||
data = resp.json()
|
resp.raise_for_status()
|
||||||
if data.get("status") != "OK":
|
data = resp.json()
|
||||||
logger.debug(f"[RoadSeq] DistanceMatrix status={data.get('status')}")
|
|
||||||
return None
|
rows = data.get("sources_to_targets") or []
|
||||||
for i, row in enumerate(data.get("rows", [])):
|
if len(rows) != n:
|
||||||
for k, el in enumerate(row.get("elements", [])):
|
logger.debug(
|
||||||
if el.get("status") == "OK":
|
f"[RoadSeq] Valhalla returned {len(rows)} rows, expected {n}"
|
||||||
dur = el.get("duration", {}).get("value")
|
)
|
||||||
if dur is not None:
|
return None
|
||||||
matrix[i][js[k]] = dur / 60.0
|
|
||||||
|
unroutable = 0
|
||||||
|
for i, row in enumerate(rows):
|
||||||
|
for el in row or []:
|
||||||
|
j = el.get("to_index")
|
||||||
|
if j is None or not (0 <= j < n):
|
||||||
|
continue
|
||||||
|
secs = el.get("time")
|
||||||
|
if secs is None:
|
||||||
|
unroutable += 1
|
||||||
|
matrix[i][j] = self._aerial_minutes(coords[i], coords[j])
|
||||||
|
else:
|
||||||
|
matrix[i][j] = secs / 60.0
|
||||||
|
|
||||||
|
pct = 100.0 * unroutable / max(1, n * n)
|
||||||
|
if pct > max_unroutable:
|
||||||
|
logger.warning(
|
||||||
|
f"[RoadSeq] {unroutable}/{n * n} pairs ({pct:.0f}%) unroutable via "
|
||||||
|
f"Valhalla - tileset likely missing this region; using aerial"
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
if unroutable:
|
||||||
|
logger.debug(
|
||||||
|
f"[RoadSeq] patched {unroutable}/{n * n} unroutable pairs with aerial"
|
||||||
|
)
|
||||||
return matrix
|
return matrix
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug(f"[RoadSeq] matrix build failed: {e}")
|
logger.debug(f"[RoadSeq] matrix build failed: {e}")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
async def _road_optimal_order(
|
async def _road_optimal_order(
|
||||||
self,
|
self,
|
||||||
start_lat: float,
|
start_lat: float,
|
||||||
@@ -140,27 +220,25 @@ class RouteOptimizer:
|
|||||||
"""
|
"""
|
||||||
Road-aware visiting order for `points`, starting from (start_lat, start_lon).
|
Road-aware visiting order for `points`, starting from (start_lat, start_lon).
|
||||||
|
|
||||||
Builds a real road travel-TIME matrix (Google Distance Matrix) and solves
|
Builds a real road travel-TIME matrix (Valhalla) and solves an OPEN TSP
|
||||||
an OPEN TSP with OR-Tools (return-to-depot edge = 0), so the sequence
|
with OR-Tools (return-to-depot edge = 0), so the sequence respects real
|
||||||
respects real road geometry/one-ways instead of straight-line distance.
|
road geometry/one-ways instead of straight-line distance. Validated on
|
||||||
Validated on live batches to cut real travel time ~5-13% vs aerial; note
|
live batches to cut real travel time ~5-13% vs aerial.
|
||||||
Google's Directions optimize:true is NOT used - it optimises a closed loop
|
|
||||||
and measured *worse* than aerial for our open delivery routes.
|
|
||||||
|
|
||||||
Returns 0-based indices into `points` in optimal order, or None to signal
|
Returns 0-based indices into `points` in optimal order, or None to signal
|
||||||
the caller to fall back to the existing aerial greedy + 2-opt.
|
the caller to fall back to the existing aerial greedy + 2-opt.
|
||||||
|
|
||||||
Safe + cheap on the live path:
|
Safe on the live path:
|
||||||
* disabled unless `routing_use_road_distance` AND a Google key is set
|
* disabled unless `routing_use_road_distance` AND VALHALLA_URL is set
|
||||||
* only for 3..`routing_road_max_stops` stops (fewer is trivial; more
|
* only for 3..`routing_road_max_stops` stops (fewer is trivial; more
|
||||||
exceeds Google's waypoint-optimize cap)
|
costs more solver time than the ordering gain is worth)
|
||||||
* result cached in Redis (default 24h) keyed by the rounded coords in
|
* result cached in Redis (default 24h) keyed by the rounded coords in
|
||||||
input order, so the returned indices always map back correctly
|
input order, so the returned indices always map back correctly
|
||||||
"""
|
"""
|
||||||
cfg = get_config()
|
cfg = get_config()
|
||||||
if not cfg.get("routing_use_road_distance", False):
|
if not cfg.get("routing_use_road_distance", False):
|
||||||
return None
|
return None
|
||||||
if not self.use_google_maps:
|
if not self.use_road_matrix:
|
||||||
return None
|
return None
|
||||||
n = len(points)
|
n = len(points)
|
||||||
if n < 3 or n > int(cfg.get("routing_road_max_stops", 25)):
|
if n < 3 or n > int(cfg.get("routing_road_max_stops", 25)):
|
||||||
@@ -362,14 +440,17 @@ class RouteOptimizer:
|
|||||||
search_params.local_search_metaheuristic = (
|
search_params.local_search_metaheuristic = (
|
||||||
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
||||||
)
|
)
|
||||||
# TSP time limit hard-capped at 2 seconds per kitchen.
|
# TSP time budget, ceiling 2 seconds per kitchen. This cap applies to
|
||||||
# search_time_limit_seconds can be tuned up to 8-10s via config, but at
|
# per-rider TSP and the per-kitchen beatmap solves.
|
||||||
# delivery scale (< 15 stops)
|
#
|
||||||
# OR-Tools finds a near-optimal solution in < 200ms. Waiting 8-10s
|
# GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES - it does not
|
||||||
# per kitchen x 3 kitchens x 4 riders = 96s of unnecessary waiting.
|
# return early once it has the optimum. A flat cap therefore burned the
|
||||||
# The VRP already has its own 3s cap. This cap applies to per-rider
|
# whole budget on trivial inputs, where PATH_CHEAPEST_ARC is done in
|
||||||
# TSP and the per-kitchen beatmap solves.
|
# under 200ms. Scale with stop count; large inputs still get the ceiling.
|
||||||
search_params.time_limit.seconds = min(self.search_time_limit_seconds, 2)
|
_cap_ms = min(self.search_time_limit_seconds, 2) * 1000
|
||||||
|
search_params.time_limit.FromMilliseconds(
|
||||||
|
max(200, min(_cap_ms, 200 * len(locations)))
|
||||||
|
)
|
||||||
|
|
||||||
solution = routing.SolveWithParameters(search_params)
|
solution = routing.SolveWithParameters(search_params)
|
||||||
|
|
||||||
@@ -742,13 +823,18 @@ class RouteOptimizer:
|
|||||||
sp.local_search_metaheuristic = (
|
sp.local_search_metaheuristic = (
|
||||||
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
||||||
)
|
)
|
||||||
# VRP time budget: cap at 3 seconds so the API response is never
|
# VRP time budget: ceiling of 3 seconds so the API response is never
|
||||||
# blocked longer than that. The x 3 multiplier caused 15-second
|
# blocked longer than that. The x 3 multiplier caused 15-second
|
||||||
# responses with the default 5-second search_time_limit.
|
# responses with the default 5-second search_time_limit.
|
||||||
# PATH_CHEAPEST_ARC typically finds a good solution in < 1 second
|
#
|
||||||
# for our typical sizes (<= 15 riders, <= 60 orders); GLS then
|
# GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES rather than
|
||||||
# improves it within the remaining budget.
|
# stopping once the optimum is found, so a flat 3s spent ~2.9s of
|
||||||
sp.time_limit.seconds = min(self.search_time_limit_seconds, 3)
|
# pure waiting on a 1-order batch (measured on live traffic).
|
||||||
|
# Scale with problem size; large batches still get the full ceiling.
|
||||||
|
_cap_ms = min(self.search_time_limit_seconds, 3) * 1000
|
||||||
|
sp.time_limit.FromMilliseconds(
|
||||||
|
max(250, min(_cap_ms, 250 * len(orders)))
|
||||||
|
)
|
||||||
|
|
||||||
solution = routing.SolveWithParameters(sp)
|
solution = routing.SolveWithParameters(sp)
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ version: "3.9"
|
|||||||
networks:
|
networks:
|
||||||
frontend:
|
frontend:
|
||||||
external: true
|
external: true
|
||||||
|
# redis + postgres-logistics live here; routes_api joins it purely to reach redis.
|
||||||
|
logistics:
|
||||||
|
external: true
|
||||||
|
|
||||||
services:
|
services:
|
||||||
routes_api:
|
routes_api:
|
||||||
@@ -14,13 +17,20 @@ services:
|
|||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
environment:
|
environment:
|
||||||
- UVICORN_WORKERS=2
|
- UVICORN_WORKERS=2
|
||||||
- REDIS_URL=redis://:${REDIS_PASSWORD}@routes_redis:6379/0
|
# Redis as discrete parts, NOT a hand-built REDIS_URL: the password contains
|
||||||
|
# reserved characters and inlining it produced redis://:pa@ss@routes_redis/...,
|
||||||
|
# which resolved to a garbage host and silently fell back to in-memory cache.
|
||||||
|
# app/services/__init__.py percent-encodes these when assembling the URL.
|
||||||
|
# Host is `redis` (its container name) on the logistics network.
|
||||||
|
- REDIS_HOST=redis
|
||||||
|
- REDIS_PORT=6379
|
||||||
|
- REDIS_PASSWORD=${REDIS_PASSWORD}
|
||||||
# Optional: Set cache TTL in seconds (default: 300 = 5 min, 86400 = 24h)
|
# Optional: Set cache TTL in seconds (default: 300 = 5 min, 86400 = 24h)
|
||||||
# Uncomment and set in .env file: REDIS_CACHE_TTL_SECONDS=86400
|
# Uncomment and set in .env file: REDIS_CACHE_TTL_SECONDS=86400
|
||||||
# - REDIS_CACHE_TTL_SECONDS=${REDIS_CACHE_TTL_SECONDS}
|
# - REDIS_CACHE_TTL_SECONDS=${REDIS_CACHE_TTL_SECONDS}
|
||||||
# Google Maps API key for accurate road distance calculation (actualkms)
|
# Self-hosted Valhalla, source of the road travel-time matrix used for
|
||||||
# Set in .env file: GOOGLE_MAPS_API_KEY=your_api_key_here
|
# road-aware stop sequencing. Unset -> sequencing falls back to aerial.
|
||||||
- GOOGLE_MAPS_API_KEY=${GOOGLE_MAPS_API_KEY}
|
- VALHALLA_URL=http://routes_valhalla:8002
|
||||||
# nearledb (read-only) — source for empirical ETA learning
|
# nearledb (read-only) — source for empirical ETA learning
|
||||||
- DB_HOST=${DB_HOST}
|
- DB_HOST=${DB_HOST}
|
||||||
- DB_PORT=${DB_PORT}
|
- DB_PORT=${DB_PORT}
|
||||||
@@ -42,3 +52,44 @@ services:
|
|||||||
- ./delivery_corrections.csv:/app/delivery_corrections.csv:ro
|
- ./delivery_corrections.csv:/app/delivery_corrections.csv:ro
|
||||||
networks:
|
networks:
|
||||||
- frontend
|
- frontend
|
||||||
|
- logistics
|
||||||
|
|
||||||
|
# Road travel-time matrix backend. Replaces the Google Distance Matrix API:
|
||||||
|
# no key, no per-request cost, no element cap, and a real `motorcycle` costing
|
||||||
|
# model instead of car durations.
|
||||||
|
#
|
||||||
|
# Deliberately NOT using tile_urls. This host has 3.8 GB RAM and runs production
|
||||||
|
# alongside it; building the 556 MB South India extract here risks the OOM killer
|
||||||
|
# taking down routes_api/postgres. Instead ./valhalla_tiles holds a pre-clipped
|
||||||
|
# Coimbatore PBF (bbox 76.60,10.70 -> 77.30,11.35, derived from the actual
|
||||||
|
# delivery coordinates in delivery_details.csv, which span only ~19 km). Valhalla
|
||||||
|
# builds from any .pbf it finds in /custom_files.
|
||||||
|
#
|
||||||
|
# To re-clip after an OSM refresh, on the host:
|
||||||
|
# osmium extract -b 76.60,10.70,77.30,11.35 <southern-zone>.osm.pbf \
|
||||||
|
# -o valhalla_tiles/coimbatore.osm.pbf
|
||||||
|
# docker compose up -d --force-recreate valhalla # with force_rebuild=True
|
||||||
|
valhalla:
|
||||||
|
image: ghcr.io/valhalla/valhalla-scripted:latest
|
||||||
|
container_name: routes_valhalla
|
||||||
|
restart: unless-stopped
|
||||||
|
environment:
|
||||||
|
- serve_tiles=True
|
||||||
|
- force_rebuild=False
|
||||||
|
- build_elevation=False
|
||||||
|
# India drives on the left; admin regions are what teach Valhalla that.
|
||||||
|
- build_admins=True
|
||||||
|
- build_time_zones=True
|
||||||
|
# Reuse built tiles on restart instead of rebuilding from the PBF.
|
||||||
|
- use_tiles_ignore_pbf=True
|
||||||
|
# Hard cap so a runaway tile build gets killed instead of the OOM killer
|
||||||
|
# picking off production containers on this 3.8 GB host.
|
||||||
|
mem_limit: 2g
|
||||||
|
volumes:
|
||||||
|
- ./valhalla_tiles:/custom_files
|
||||||
|
# Loopback-only: the API reaches Valhalla over the docker network, so this
|
||||||
|
# is purely for local probing. Valhalla has no auth — do not bind 0.0.0.0.
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8002:8002"
|
||||||
|
networks:
|
||||||
|
- frontend
|
||||||
|
|||||||
@@ -11,3 +11,4 @@ ortools
|
|||||||
python-dateutil
|
python-dateutil
|
||||||
faiss-cpu
|
faiss-cpu
|
||||||
psycopg2-binary
|
psycopg2-binary
|
||||||
|
redis
|
||||||
|
|||||||
Reference in New Issue
Block a user