sync: capture the production server's code, which was never committed
/root/Routes-api on 31.97.228.132 is not a git repository. Work had been done directly on the box and existed nowhere else -- a single rm -rf from being lost, and impossible to review or roll back. Deploying the previous HEAD over it would have silently reverted all of this. Most visibly the Valhalla road-backend probe in main.py, whose own comment explains why it exists: road sequencing degrades to aerial silently by design, so an unreachable backend stays invisible, "which is exactly how the expired Google key went unnoticed". Overwriting it would have reintroduced precisely the failure it was written to catch, and the service would have kept answering 200 throughout. The server had also moved from Google Maps to Valhalla for road distance (VALHALLA_URL, road_backend_status, +190 lines in route_optimizer), extended docker-compose from 44 to 95 lines, and changed rider fetching, health, dynamic config and the cache layer. Only 10 files differ in substance. The other 27 that appeared to differ were CRLF-vs-LF noise -- the server writes CRLF -- and are normalised to LF here rather than committed as spurious whole-file rewrites. Committed as-is, before any change of mine, so the diff that follows is reviewable against what is actually running. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -38,6 +38,9 @@ DEFAULTS: Dict[str, Any] = {
|
|||||||
"emergency_load_penalty": 3.0, # km penalty per order in emergency assign
|
"emergency_load_penalty": 3.0, # km penalty per order in emergency assign
|
||||||
# RouteOptimizer
|
# RouteOptimizer
|
||||||
"search_time_limit_seconds": 5,
|
"search_time_limit_seconds": 5,
|
||||||
|
# Active-rider roster cache. The upstream getriderlogs call measured ~6s on
|
||||||
|
# live traffic. Short TTL so on/off-duty changes surface quickly; 0 = off.
|
||||||
|
"rider_roster_cache_ttl_seconds": 30,
|
||||||
"avg_speed_kmh": 18.0,
|
"avg_speed_kmh": 18.0,
|
||||||
"road_factor": 1.3,
|
"road_factor": 1.3,
|
||||||
# ClusteringService
|
# ClusteringService
|
||||||
@@ -56,12 +59,20 @@ DEFAULTS: Dict[str, Any] = {
|
|||||||
"eta_history_days": 14, # rolling window pulled from nearledb
|
"eta_history_days": 14, # rolling window pulled from nearledb
|
||||||
"eta_stat": "median", # "median" or "p75" (p75 = more conservative)
|
"eta_stat": "median", # "median" or "p75" (p75 = more conservative)
|
||||||
"eta_sync_interval_hours": 6, # autonomous background sync cadence
|
"eta_sync_interval_hours": 6, # autonomous background sync cadence
|
||||||
# Road-aware sequencing (Phase 2). OFF by default: enabling adds a Google
|
# Road-aware sequencing (Phase 2). Backed by self-hosted Valhalla, so there is
|
||||||
# Directions call (cost + latency) to the route hot path. Results are cached.
|
# no per-request cost — only local latency. Still OFF by default and left to
|
||||||
# Only the *visiting order* changes; step/ETA metrics stay aerial-based.
|
# the agent to enable on measured gain. Only the *visiting order* changes;
|
||||||
|
# step/ETA metrics stay aerial-based.
|
||||||
"routing_use_road_distance": False, # AGENT-MANAGED (see routing_auto_manage)
|
"routing_use_road_distance": False, # AGENT-MANAGED (see routing_auto_manage)
|
||||||
"routing_road_cache_ttl_seconds": 86400, # road geometry is stable; cache 24h
|
"routing_road_cache_ttl_seconds": 86400, # road geometry is stable; cache 24h
|
||||||
"routing_road_max_stops": 25, # Google distance-matrix practical cap
|
"routing_road_max_stops": 25, # beyond this the solver time outweighs
|
||||||
|
# the ordering gain (not a backend cap)
|
||||||
|
# Valhalla matrix backend. "motorcycle" models the lane access and one-way
|
||||||
|
# behaviour our riders actually have; "auto" would overstate their travel time.
|
||||||
|
"routing_valhalla_costing": "motorcycle",
|
||||||
|
"routing_matrix_timeout_seconds": 15.0,
|
||||||
|
"routing_matrix_max_unroutable_pct": 20.0, # above this -> tiles likely missing
|
||||||
|
# this region, fall back to aerial
|
||||||
# Autonomous road-sequencing decision agent: measures road-vs-aerial travel
|
# Autonomous road-sequencing decision agent: measures road-vs-aerial travel
|
||||||
# time on real batches and flips routing_use_road_distance on its own.
|
# time on real batches and flips routing_use_road_distance on its own.
|
||||||
"routing_auto_manage": True, # False -> humans own the flag
|
"routing_auto_manage": True, # False -> humans own the flag
|
||||||
|
|||||||
26
app/main.py
26
app/main.py
@@ -6,7 +6,7 @@ import sys
|
|||||||
import time
|
import time
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
|
|
||||||
# Load .env (DB_*, REDIS_*, GOOGLE_MAPS_API_KEY) into the environment early,
|
# Load .env (DB_*, REDIS_*, VALHALLA_URL) into the environment early,
|
||||||
# before any service reads os.getenv at import time.
|
# before any service reads os.getenv at import time.
|
||||||
try:
|
try:
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
@@ -28,7 +28,7 @@ from app.core.exception_handlers import (
|
|||||||
)
|
)
|
||||||
from app.core.exceptions import APIException
|
from app.core.exceptions import APIException
|
||||||
from app.middleware.request_id import RequestIDMiddleware
|
from app.middleware.request_id import RequestIDMiddleware
|
||||||
from app.routes import cache_router, health_router, ml_router, ml_web_router, optimization_router, batch_analytics_router, riders_router, doormile_router
|
from app.routes import cache_router, health_router, ml_router, ml_web_router, optimization_router, batch_analytics_router, riders_router
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Logging
|
# Logging
|
||||||
@@ -111,6 +111,27 @@ async def lifespan(app: FastAPI):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning(f"[RoadAgent] Decision agent start failed (non-fatal): {e}")
|
logger.warning(f"[RoadAgent] Decision agent start failed (non-fatal): {e}")
|
||||||
|
|
||||||
|
# Probe the Valhalla matrix backend once. Road sequencing degrades to aerial
|
||||||
|
# silently by design, so without this an unreachable or still-building backend
|
||||||
|
# stays invisible — which is exactly how the expired Google key went unnoticed.
|
||||||
|
try:
|
||||||
|
from app.services.routing.route_optimizer import road_backend_status
|
||||||
|
_rb = await road_backend_status()
|
||||||
|
if not _rb.get("configured"):
|
||||||
|
logger.warning("[RoadMatrix] VALHALLA_URL not set - aerial sequencing only.")
|
||||||
|
elif _rb.get("reachable"):
|
||||||
|
logger.info(
|
||||||
|
f"[RoadMatrix] Valhalla ready at {_rb.get('url')} "
|
||||||
|
f"(version={_rb.get('version')}, tiles={_rb.get('tileset_last_modified')})"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"[RoadMatrix] Valhalla at {_rb.get('url')} unreachable "
|
||||||
|
f"({_rb.get('detail')}) - aerial fallback until it answers."
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"[RoadMatrix] Backend probe failed (non-fatal): {e}")
|
||||||
|
|
||||||
logger.info("[OK] Application ready.")
|
logger.info("[OK] Application ready.")
|
||||||
yield
|
yield
|
||||||
logger.info("[STOP] Route Optimization API shutting down.")
|
logger.info("[STOP] Route Optimization API shutting down.")
|
||||||
@@ -162,7 +183,6 @@ app.include_router(ml_router)
|
|||||||
app.include_router(ml_web_router)
|
app.include_router(ml_web_router)
|
||||||
app.include_router(batch_analytics_router)
|
app.include_router(batch_analytics_router)
|
||||||
app.include_router(riders_router)
|
app.include_router(riders_router)
|
||||||
app.include_router(doormile_router)
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/", tags=["Root"])
|
@app.get("/", tags=["Root"])
|
||||||
|
|||||||
@@ -6,7 +6,6 @@ from .cache import router as cache_router
|
|||||||
from .ml_admin import router as ml_router, web_router as ml_web_router
|
from .ml_admin import router as ml_router, web_router as ml_web_router
|
||||||
from .batch_analytics import router as batch_analytics_router
|
from .batch_analytics import router as batch_analytics_router
|
||||||
from .riders import router as riders_router
|
from .riders import router as riders_router
|
||||||
from .doormile import router as doormile_router
|
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"optimization_router",
|
"optimization_router",
|
||||||
@@ -16,5 +15,4 @@ __all__ = [
|
|||||||
"ml_web_router",
|
"ml_web_router",
|
||||||
"batch_analytics_router",
|
"batch_analytics_router",
|
||||||
"riders_router",
|
"riders_router",
|
||||||
"doormile_router",
|
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -84,6 +84,26 @@ async def readiness_check(request: Request):
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/road-backend")
|
||||||
|
async def road_backend_check(request: Request):
|
||||||
|
"""
|
||||||
|
Status of the Valhalla road-matrix backend used for road-aware stop sequencing.
|
||||||
|
|
||||||
|
Deliberately separate from /ready: an unreachable backend is a degradation,
|
||||||
|
not an outage — routing falls back to aerial ordering and keeps serving. This
|
||||||
|
endpoint is how you notice the degradation.
|
||||||
|
"""
|
||||||
|
from app.services.routing.route_optimizer import road_backend_status
|
||||||
|
|
||||||
|
status = await road_backend_status()
|
||||||
|
return {
|
||||||
|
**status,
|
||||||
|
"sequencing_mode": "road" if status.get("reachable") else "aerial",
|
||||||
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
||||||
|
"request_id": getattr(request.state, "request_id", None),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@router.get("/live")
|
@router.get("/live")
|
||||||
async def liveness_check(request: Request):
|
async def liveness_check(request: Request):
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -15,6 +15,31 @@ except Exception: # pragma: no cover
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _redis_url_from_parts() -> Optional[str]:
|
||||||
|
"""
|
||||||
|
Build a redis:// URL from discrete env vars, percent-encoding the credentials.
|
||||||
|
|
||||||
|
A hand-written REDIS_URL breaks the moment the password contains a reserved
|
||||||
|
character: an '@' splits the authority section early, so the client silently
|
||||||
|
dials a garbage host instead of failing loudly. Assembling the URL here with
|
||||||
|
quote() removes that whole class of bug - callers set REDIS_HOST/REDIS_PASSWORD
|
||||||
|
and never have to think about escaping.
|
||||||
|
|
||||||
|
Returns None when REDIS_HOST is unset, so an explicit REDIS_URL still wins.
|
||||||
|
"""
|
||||||
|
host = os.getenv("REDIS_HOST")
|
||||||
|
if not host:
|
||||||
|
return None
|
||||||
|
from urllib.parse import quote
|
||||||
|
|
||||||
|
user = os.getenv("REDIS_USERNAME", "")
|
||||||
|
pwd = os.getenv("REDIS_PASSWORD", "")
|
||||||
|
port = os.getenv("REDIS_PORT", "6379")
|
||||||
|
db = os.getenv("REDIS_DB", "0")
|
||||||
|
auth = f"{quote(user, safe='')}:{quote(pwd, safe='')}@" if pwd else ""
|
||||||
|
return f"redis://{auth}{host}:{port}/{db}"
|
||||||
|
|
||||||
|
|
||||||
class RedisCache:
|
class RedisCache:
|
||||||
"""Lightweight Redis cache wrapper with graceful in-memory fallback."""
|
"""Lightweight Redis cache wrapper with graceful in-memory fallback."""
|
||||||
|
|
||||||
@@ -33,9 +58,10 @@ class RedisCache:
|
|||||||
self._client = None
|
self._client = None
|
||||||
self._stats = {"hits": 0, "misses": 0, "sets": 0}
|
self._stats = {"hits": 0, "misses": 0, "sets": 0}
|
||||||
|
|
||||||
url = os.getenv(url_env)
|
url = os.getenv(url_env) or _redis_url_from_parts()
|
||||||
if not url or redis is None:
|
if not url or redis is None:
|
||||||
logger.warning("Redis not configured or client unavailable; falling back to local thread-safe in-memory cache")
|
_why = "redis client not installed" if redis is None else "no REDIS_URL/REDIS_HOST"
|
||||||
|
logger.warning(f"Redis not configured or client unavailable ({_why}); falling back to local thread-safe in-memory cache")
|
||||||
return
|
return
|
||||||
try:
|
try:
|
||||||
self._client = redis.Redis.from_url(url, decode_responses=True)
|
self._client = redis.Redis.from_url(url, decode_responses=True)
|
||||||
|
|||||||
@@ -6,13 +6,45 @@ from typing import List, Dict, Any
|
|||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
def _roster_cache_ttl() -> int:
|
||||||
|
"""Roster cache TTL in seconds. 0 disables caching entirely."""
|
||||||
|
try:
|
||||||
|
from app.config.dynamic_config import get_config
|
||||||
|
return int(get_config().get("rider_roster_cache_ttl_seconds", 30))
|
||||||
|
except Exception:
|
||||||
|
return 30
|
||||||
|
|
||||||
|
|
||||||
async def fetch_active_riders() -> List[Dict[str, Any]]:
|
async def fetch_active_riders() -> List[Dict[str, Any]]:
|
||||||
"""
|
"""
|
||||||
Fetch active rider logs from the external API for the current date.
|
Fetch active rider logs from the external API for the current date.
|
||||||
Returns a list of rider log dictionaries.
|
Returns a list of rider log dictionaries.
|
||||||
|
|
||||||
|
This upstream call measured ~6s on live traffic and was the single largest
|
||||||
|
component of riderassign latency, so successful responses are cached for
|
||||||
|
`rider_roster_cache_ttl_seconds`. The TTL is deliberately short: a rider
|
||||||
|
going on or off duty must become visible quickly, and the absent-rider
|
||||||
|
validation in the assignment path checks against this roster.
|
||||||
|
|
||||||
|
Only non-empty successes are cached. Caching an empty result or a failure
|
||||||
|
would turn a brief upstream blip into a full-TTL outage where every request
|
||||||
|
sees zero riders.
|
||||||
"""
|
"""
|
||||||
|
today_str = datetime.now().strftime("%Y-%m-%d")
|
||||||
|
ttl = _roster_cache_ttl()
|
||||||
|
cache_key = f"riders:active:{today_str}"
|
||||||
|
|
||||||
|
if ttl > 0:
|
||||||
|
try:
|
||||||
|
from app.services import cache as _cache
|
||||||
|
cached = _cache.get_json(cache_key)
|
||||||
|
if isinstance(cached, list) and cached:
|
||||||
|
logger.debug(f"[Riders] roster cache hit ({len(cached)} riders)")
|
||||||
|
return cached
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
try:
|
try:
|
||||||
today_str = datetime.now().strftime("%Y-%m-%d")
|
|
||||||
url = "https://jupiter.nearle.app/live/api/v2/partners/getriderlogs/"
|
url = "https://jupiter.nearle.app/live/api/v2/partners/getriderlogs/"
|
||||||
params = {
|
params = {
|
||||||
"applocationid": 1,
|
"applocationid": 1,
|
||||||
@@ -32,7 +64,14 @@ async def fetch_active_riders() -> List[Dict[str, Any]]:
|
|||||||
# The user's example showed "onduty": 1. We might want to filter by that.
|
# The user's example showed "onduty": 1. We might want to filter by that.
|
||||||
# For now, returning all logs, filtering can happen in assignment logic or here.
|
# For now, returning all logs, filtering can happen in assignment logic or here.
|
||||||
# Let's return the raw list as requested, filtering logic will be applied during assignment.
|
# Let's return the raw list as requested, filtering logic will be applied during assignment.
|
||||||
return data.get("details", [])
|
riders = data.get("details", [])
|
||||||
|
if riders and ttl > 0:
|
||||||
|
try:
|
||||||
|
from app.services import cache as _cache
|
||||||
|
_cache.set_json(cache_key, riders, ttl_seconds=ttl)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return riders
|
||||||
|
|
||||||
logger.warning(f"Fetch active riders returned no details: {data}")
|
logger.warning(f"Fetch active riders returned no details: {data}")
|
||||||
return []
|
return []
|
||||||
|
|||||||
@@ -61,8 +61,8 @@ class RoadSequencingAgent:
|
|||||||
from app.core.arrow_utils import calculate_haversine_matrix_vectorized
|
from app.core.arrow_utils import calculate_haversine_matrix_vectorized
|
||||||
|
|
||||||
opt = self._optimizer()
|
opt = self._optimizer()
|
||||||
if not opt.use_google_maps:
|
if not opt.use_road_matrix:
|
||||||
return {"evaluated": 0, "reason": "no_google_key", "mean_gain_pct": 0.0}
|
return {"evaluated": 0, "reason": "no_matrix_backend", "mean_gain_pct": 0.0}
|
||||||
|
|
||||||
batches = get_delivery_history_service().sample_batches(
|
batches = get_delivery_history_service().sample_batches(
|
||||||
days=days, limit=sample_batches
|
days=days, limit=sample_batches
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ ALGORITHM: TSP / VRP with Google OR-Tools
|
|||||||
|
|
||||||
FEATURES:
|
FEATURES:
|
||||||
- Automatic outlier detection and coordinate correction
|
- Automatic outlier detection and coordinate correction
|
||||||
- Hybrid distance calculation (Google Maps + Haversine fallback)
|
- Hybrid distance calculation (self-hosted Valhalla road matrix + Haversine fallback)
|
||||||
- Robust error handling for invalid inputs
|
- Robust error handling for invalid inputs
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -37,6 +37,37 @@ except ImportError:
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def road_backend_status() -> Dict[str, Any]:
|
||||||
|
"""
|
||||||
|
Probe the Valhalla matrix backend.
|
||||||
|
|
||||||
|
Module-level (not a RouteOptimizer method) so startup and health checks can
|
||||||
|
call it without paying for a full optimizer construction, which pulls in the
|
||||||
|
empirical ETA calculator and its history load.
|
||||||
|
"""
|
||||||
|
url = os.getenv("VALHALLA_URL", "").strip().rstrip("/")
|
||||||
|
if not url:
|
||||||
|
return {
|
||||||
|
"configured": False,
|
||||||
|
"reachable": False,
|
||||||
|
"detail": "VALHALLA_URL not set - road sequencing disabled, aerial only",
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||||
|
resp = await client.get(f"{url}/status")
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
return {
|
||||||
|
"configured": True,
|
||||||
|
"reachable": True,
|
||||||
|
"url": url,
|
||||||
|
"version": data.get("version"),
|
||||||
|
"tileset_last_modified": data.get("tileset_last_modified"),
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
return {"configured": True, "reachable": False, "url": url, "detail": str(e)}
|
||||||
|
|
||||||
|
|
||||||
class RouteOptimizer:
|
class RouteOptimizer:
|
||||||
"""Route optimization using Google OR-Tools (Async)."""
|
"""Route optimization using Google OR-Tools (Async)."""
|
||||||
|
|
||||||
@@ -60,9 +91,11 @@ class RouteOptimizer:
|
|||||||
# Road factor (haversine -> road distance multiplier, ML-tuned)
|
# Road factor (haversine -> road distance multiplier, ML-tuned)
|
||||||
self.road_factor = float(_cfg.get("road_factor"))
|
self.road_factor = float(_cfg.get("road_factor"))
|
||||||
|
|
||||||
# Google Maps API settings
|
# Road travel-time matrix backend: self-hosted Valhalla (no per-request
|
||||||
self.google_maps_api_key = os.getenv("GOOGLE_MAPS_API_KEY", "")
|
# cost, no rate limit). Unset VALHALLA_URL -> road sequencing stays off
|
||||||
self.use_google_maps = bool(self.google_maps_api_key)
|
# and every caller falls back to aerial ordering.
|
||||||
|
self.valhalla_url = os.getenv("VALHALLA_URL", "").strip().rstrip("/")
|
||||||
|
self.use_road_matrix = bool(self.valhalla_url)
|
||||||
|
|
||||||
# Solver time limit (ML-tuned)
|
# Solver time limit (ML-tuned)
|
||||||
self.search_time_limit_seconds = int(_cfg.get("search_time_limit_seconds"))
|
self.search_time_limit_seconds = int(_cfg.get("search_time_limit_seconds"))
|
||||||
@@ -90,47 +123,94 @@ class RouteOptimizer:
|
|||||||
# ROAD-AWARE VISITING ORDER (Phase 2 - opt-in, cached)
|
# ROAD-AWARE VISITING ORDER (Phase 2 - opt-in, cached)
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _aerial_minutes(
|
||||||
|
self, a: Tuple[float, float], b: Tuple[float, float]
|
||||||
|
) -> float:
|
||||||
|
"""Haversine travel-time estimate (minutes), used to patch unroutable pairs."""
|
||||||
|
km = self.haversine_distance(a[0], a[1], b[0], b[1]) * self.road_factor
|
||||||
|
return (km / (self.avg_speed_kmh or 20.0)) * 60.0
|
||||||
|
|
||||||
async def _road_duration_matrix(
|
async def _road_duration_matrix(
|
||||||
self, coords: _List[Tuple[float, float]]
|
self, coords: _List[Tuple[float, float]]
|
||||||
) -> Optional[_List[_List[float]]]:
|
) -> Optional[_List[_List[float]]]:
|
||||||
"""
|
"""
|
||||||
Full NxN road travel-TIME matrix (minutes) via Google Distance Matrix.
|
Full NxN road travel-TIME matrix (minutes) via self-hosted Valhalla.
|
||||||
|
|
||||||
|
One /sources_to_targets call returns the whole matrix, so unlike the old
|
||||||
|
Google Distance Matrix path there is no per-request element cap and no
|
||||||
|
chunking. Costing defaults to `motorcycle`, which models the lane access
|
||||||
|
and one-way behaviour our riders actually have — a car matrix systematically
|
||||||
|
overstates their travel time.
|
||||||
|
|
||||||
|
Valhalla reports an unroutable pair as time=null. Those are patched with an
|
||||||
|
aerial estimate rather than left at 0, because a 0-cost edge would look
|
||||||
|
free to the TSP and pull the whole sequence through it. If too many pairs
|
||||||
|
are unroutable the tileset probably doesn't cover this region, so we bail
|
||||||
|
to aerial entirely.
|
||||||
|
|
||||||
Chunks destinations to respect Google's ~100-elements-per-request limit.
|
|
||||||
Returns None on any failure so the caller falls back to aerial ordering.
|
Returns None on any failure so the caller falls back to aerial ordering.
|
||||||
"""
|
"""
|
||||||
if not self.use_google_maps:
|
if not self.use_road_matrix:
|
||||||
return None
|
return None
|
||||||
|
cfg = get_config()
|
||||||
|
costing = str(cfg.get("routing_valhalla_costing", "motorcycle"))
|
||||||
|
timeout = float(cfg.get("routing_matrix_timeout_seconds", 15.0))
|
||||||
|
max_unroutable = float(cfg.get("routing_matrix_max_unroutable_pct", 20.0))
|
||||||
|
|
||||||
n = len(coords)
|
n = len(coords)
|
||||||
origins = "|".join(f"{la},{lo}" for la, lo in coords)
|
locations = [{"lat": float(la), "lon": float(lo)} for la, lo in coords]
|
||||||
matrix = [[0.0] * n for _ in range(n)]
|
matrix = [[0.0] * n for _ in range(n)]
|
||||||
dest_chunk = max(1, 100 // max(1, n))
|
|
||||||
try:
|
try:
|
||||||
async with httpx.AsyncClient(timeout=15.0) as client:
|
async with httpx.AsyncClient(timeout=timeout) as client:
|
||||||
for j0 in range(0, n, dest_chunk):
|
resp = await client.post(
|
||||||
js = list(range(j0, min(j0 + dest_chunk, n)))
|
f"{self.valhalla_url}/sources_to_targets",
|
||||||
dests = "|".join(f"{coords[j][0]},{coords[j][1]}" for j in js)
|
json={
|
||||||
resp = await client.get(
|
"sources": locations,
|
||||||
"https://maps.googleapis.com/maps/api/distancematrix/json",
|
"targets": locations,
|
||||||
params={"origins": origins, "destinations": dests,
|
"costing": costing,
|
||||||
"key": self.google_maps_api_key, "units": "metric"},
|
"units": "km",
|
||||||
)
|
},
|
||||||
resp.raise_for_status()
|
)
|
||||||
data = resp.json()
|
resp.raise_for_status()
|
||||||
if data.get("status") != "OK":
|
data = resp.json()
|
||||||
logger.debug(f"[RoadSeq] DistanceMatrix status={data.get('status')}")
|
|
||||||
return None
|
rows = data.get("sources_to_targets") or []
|
||||||
for i, row in enumerate(data.get("rows", [])):
|
if len(rows) != n:
|
||||||
for k, el in enumerate(row.get("elements", [])):
|
logger.debug(
|
||||||
if el.get("status") == "OK":
|
f"[RoadSeq] Valhalla returned {len(rows)} rows, expected {n}"
|
||||||
dur = el.get("duration", {}).get("value")
|
)
|
||||||
if dur is not None:
|
return None
|
||||||
matrix[i][js[k]] = dur / 60.0
|
|
||||||
|
unroutable = 0
|
||||||
|
for i, row in enumerate(rows):
|
||||||
|
for el in row or []:
|
||||||
|
j = el.get("to_index")
|
||||||
|
if j is None or not (0 <= j < n):
|
||||||
|
continue
|
||||||
|
secs = el.get("time")
|
||||||
|
if secs is None:
|
||||||
|
unroutable += 1
|
||||||
|
matrix[i][j] = self._aerial_minutes(coords[i], coords[j])
|
||||||
|
else:
|
||||||
|
matrix[i][j] = secs / 60.0
|
||||||
|
|
||||||
|
pct = 100.0 * unroutable / max(1, n * n)
|
||||||
|
if pct > max_unroutable:
|
||||||
|
logger.warning(
|
||||||
|
f"[RoadSeq] {unroutable}/{n * n} pairs ({pct:.0f}%) unroutable via "
|
||||||
|
f"Valhalla - tileset likely missing this region; using aerial"
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
if unroutable:
|
||||||
|
logger.debug(
|
||||||
|
f"[RoadSeq] patched {unroutable}/{n * n} unroutable pairs with aerial"
|
||||||
|
)
|
||||||
return matrix
|
return matrix
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug(f"[RoadSeq] matrix build failed: {e}")
|
logger.debug(f"[RoadSeq] matrix build failed: {e}")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
async def _road_optimal_order(
|
async def _road_optimal_order(
|
||||||
self,
|
self,
|
||||||
start_lat: float,
|
start_lat: float,
|
||||||
@@ -140,27 +220,25 @@ class RouteOptimizer:
|
|||||||
"""
|
"""
|
||||||
Road-aware visiting order for `points`, starting from (start_lat, start_lon).
|
Road-aware visiting order for `points`, starting from (start_lat, start_lon).
|
||||||
|
|
||||||
Builds a real road travel-TIME matrix (Google Distance Matrix) and solves
|
Builds a real road travel-TIME matrix (Valhalla) and solves an OPEN TSP
|
||||||
an OPEN TSP with OR-Tools (return-to-depot edge = 0), so the sequence
|
with OR-Tools (return-to-depot edge = 0), so the sequence respects real
|
||||||
respects real road geometry/one-ways instead of straight-line distance.
|
road geometry/one-ways instead of straight-line distance. Validated on
|
||||||
Validated on live batches to cut real travel time ~5-13% vs aerial; note
|
live batches to cut real travel time ~5-13% vs aerial.
|
||||||
Google's Directions optimize:true is NOT used - it optimises a closed loop
|
|
||||||
and measured *worse* than aerial for our open delivery routes.
|
|
||||||
|
|
||||||
Returns 0-based indices into `points` in optimal order, or None to signal
|
Returns 0-based indices into `points` in optimal order, or None to signal
|
||||||
the caller to fall back to the existing aerial greedy + 2-opt.
|
the caller to fall back to the existing aerial greedy + 2-opt.
|
||||||
|
|
||||||
Safe + cheap on the live path:
|
Safe on the live path:
|
||||||
* disabled unless `routing_use_road_distance` AND a Google key is set
|
* disabled unless `routing_use_road_distance` AND VALHALLA_URL is set
|
||||||
* only for 3..`routing_road_max_stops` stops (fewer is trivial; more
|
* only for 3..`routing_road_max_stops` stops (fewer is trivial; more
|
||||||
exceeds Google's waypoint-optimize cap)
|
costs more solver time than the ordering gain is worth)
|
||||||
* result cached in Redis (default 24h) keyed by the rounded coords in
|
* result cached in Redis (default 24h) keyed by the rounded coords in
|
||||||
input order, so the returned indices always map back correctly
|
input order, so the returned indices always map back correctly
|
||||||
"""
|
"""
|
||||||
cfg = get_config()
|
cfg = get_config()
|
||||||
if not cfg.get("routing_use_road_distance", False):
|
if not cfg.get("routing_use_road_distance", False):
|
||||||
return None
|
return None
|
||||||
if not self.use_google_maps:
|
if not self.use_road_matrix:
|
||||||
return None
|
return None
|
||||||
n = len(points)
|
n = len(points)
|
||||||
if n < 3 or n > int(cfg.get("routing_road_max_stops", 25)):
|
if n < 3 or n > int(cfg.get("routing_road_max_stops", 25)):
|
||||||
@@ -362,14 +440,17 @@ class RouteOptimizer:
|
|||||||
search_params.local_search_metaheuristic = (
|
search_params.local_search_metaheuristic = (
|
||||||
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
||||||
)
|
)
|
||||||
# TSP time limit hard-capped at 2 seconds per kitchen.
|
# TSP time budget, ceiling 2 seconds per kitchen. This cap applies to
|
||||||
# search_time_limit_seconds can be tuned up to 8-10s via config, but at
|
# per-rider TSP and the per-kitchen beatmap solves.
|
||||||
# delivery scale (< 15 stops)
|
#
|
||||||
# OR-Tools finds a near-optimal solution in < 200ms. Waiting 8-10s
|
# GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES - it does not
|
||||||
# per kitchen x 3 kitchens x 4 riders = 96s of unnecessary waiting.
|
# return early once it has the optimum. A flat cap therefore burned the
|
||||||
# The VRP already has its own 3s cap. This cap applies to per-rider
|
# whole budget on trivial inputs, where PATH_CHEAPEST_ARC is done in
|
||||||
# TSP and the per-kitchen beatmap solves.
|
# under 200ms. Scale with stop count; large inputs still get the ceiling.
|
||||||
search_params.time_limit.seconds = min(self.search_time_limit_seconds, 2)
|
_cap_ms = min(self.search_time_limit_seconds, 2) * 1000
|
||||||
|
search_params.time_limit.FromMilliseconds(
|
||||||
|
max(200, min(_cap_ms, 200 * len(locations)))
|
||||||
|
)
|
||||||
|
|
||||||
solution = routing.SolveWithParameters(search_params)
|
solution = routing.SolveWithParameters(search_params)
|
||||||
|
|
||||||
@@ -742,13 +823,18 @@ class RouteOptimizer:
|
|||||||
sp.local_search_metaheuristic = (
|
sp.local_search_metaheuristic = (
|
||||||
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH
|
||||||
)
|
)
|
||||||
# VRP time budget: cap at 3 seconds so the API response is never
|
# VRP time budget: ceiling of 3 seconds so the API response is never
|
||||||
# blocked longer than that. The x 3 multiplier caused 15-second
|
# blocked longer than that. The x 3 multiplier caused 15-second
|
||||||
# responses with the default 5-second search_time_limit.
|
# responses with the default 5-second search_time_limit.
|
||||||
# PATH_CHEAPEST_ARC typically finds a good solution in < 1 second
|
#
|
||||||
# for our typical sizes (<= 15 riders, <= 60 orders); GLS then
|
# GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES rather than
|
||||||
# improves it within the remaining budget.
|
# stopping once the optimum is found, so a flat 3s spent ~2.9s of
|
||||||
sp.time_limit.seconds = min(self.search_time_limit_seconds, 3)
|
# pure waiting on a 1-order batch (measured on live traffic).
|
||||||
|
# Scale with problem size; large batches still get the full ceiling.
|
||||||
|
_cap_ms = min(self.search_time_limit_seconds, 3) * 1000
|
||||||
|
sp.time_limit.FromMilliseconds(
|
||||||
|
max(250, min(_cap_ms, 250 * len(orders)))
|
||||||
|
)
|
||||||
|
|
||||||
solution = routing.SolveWithParameters(sp)
|
solution = routing.SolveWithParameters(sp)
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ version: "3.9"
|
|||||||
networks:
|
networks:
|
||||||
frontend:
|
frontend:
|
||||||
external: true
|
external: true
|
||||||
|
# redis + postgres-logistics live here; routes_api joins it purely to reach redis.
|
||||||
|
logistics:
|
||||||
|
external: true
|
||||||
|
|
||||||
services:
|
services:
|
||||||
routes_api:
|
routes_api:
|
||||||
@@ -14,13 +17,20 @@ services:
|
|||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
environment:
|
environment:
|
||||||
- UVICORN_WORKERS=2
|
- UVICORN_WORKERS=2
|
||||||
- REDIS_URL=redis://:${REDIS_PASSWORD}@routes_redis:6379/0
|
# Redis as discrete parts, NOT a hand-built REDIS_URL: the password contains
|
||||||
|
# reserved characters and inlining it produced redis://:pa@ss@routes_redis/...,
|
||||||
|
# which resolved to a garbage host and silently fell back to in-memory cache.
|
||||||
|
# app/services/__init__.py percent-encodes these when assembling the URL.
|
||||||
|
# Host is `redis` (its container name) on the logistics network.
|
||||||
|
- REDIS_HOST=redis
|
||||||
|
- REDIS_PORT=6379
|
||||||
|
- REDIS_PASSWORD=${REDIS_PASSWORD}
|
||||||
# Optional: Set cache TTL in seconds (default: 300 = 5 min, 86400 = 24h)
|
# Optional: Set cache TTL in seconds (default: 300 = 5 min, 86400 = 24h)
|
||||||
# Uncomment and set in .env file: REDIS_CACHE_TTL_SECONDS=86400
|
# Uncomment and set in .env file: REDIS_CACHE_TTL_SECONDS=86400
|
||||||
# - REDIS_CACHE_TTL_SECONDS=${REDIS_CACHE_TTL_SECONDS}
|
# - REDIS_CACHE_TTL_SECONDS=${REDIS_CACHE_TTL_SECONDS}
|
||||||
# Google Maps API key for accurate road distance calculation (actualkms)
|
# Self-hosted Valhalla, source of the road travel-time matrix used for
|
||||||
# Set in .env file: GOOGLE_MAPS_API_KEY=your_api_key_here
|
# road-aware stop sequencing. Unset -> sequencing falls back to aerial.
|
||||||
- GOOGLE_MAPS_API_KEY=${GOOGLE_MAPS_API_KEY}
|
- VALHALLA_URL=http://routes_valhalla:8002
|
||||||
# nearledb (read-only) — source for empirical ETA learning
|
# nearledb (read-only) — source for empirical ETA learning
|
||||||
- DB_HOST=${DB_HOST}
|
- DB_HOST=${DB_HOST}
|
||||||
- DB_PORT=${DB_PORT}
|
- DB_PORT=${DB_PORT}
|
||||||
@@ -42,3 +52,44 @@ services:
|
|||||||
- ./delivery_corrections.csv:/app/delivery_corrections.csv:ro
|
- ./delivery_corrections.csv:/app/delivery_corrections.csv:ro
|
||||||
networks:
|
networks:
|
||||||
- frontend
|
- frontend
|
||||||
|
- logistics
|
||||||
|
|
||||||
|
# Road travel-time matrix backend. Replaces the Google Distance Matrix API:
|
||||||
|
# no key, no per-request cost, no element cap, and a real `motorcycle` costing
|
||||||
|
# model instead of car durations.
|
||||||
|
#
|
||||||
|
# Deliberately NOT using tile_urls. This host has 3.8 GB RAM and runs production
|
||||||
|
# alongside it; building the 556 MB South India extract here risks the OOM killer
|
||||||
|
# taking down routes_api/postgres. Instead ./valhalla_tiles holds a pre-clipped
|
||||||
|
# Coimbatore PBF (bbox 76.60,10.70 -> 77.30,11.35, derived from the actual
|
||||||
|
# delivery coordinates in delivery_details.csv, which span only ~19 km). Valhalla
|
||||||
|
# builds from any .pbf it finds in /custom_files.
|
||||||
|
#
|
||||||
|
# To re-clip after an OSM refresh, on the host:
|
||||||
|
# osmium extract -b 76.60,10.70,77.30,11.35 <southern-zone>.osm.pbf \
|
||||||
|
# -o valhalla_tiles/coimbatore.osm.pbf
|
||||||
|
# docker compose up -d --force-recreate valhalla # with force_rebuild=True
|
||||||
|
valhalla:
|
||||||
|
image: ghcr.io/valhalla/valhalla-scripted:latest
|
||||||
|
container_name: routes_valhalla
|
||||||
|
restart: unless-stopped
|
||||||
|
environment:
|
||||||
|
- serve_tiles=True
|
||||||
|
- force_rebuild=False
|
||||||
|
- build_elevation=False
|
||||||
|
# India drives on the left; admin regions are what teach Valhalla that.
|
||||||
|
- build_admins=True
|
||||||
|
- build_time_zones=True
|
||||||
|
# Reuse built tiles on restart instead of rebuilding from the PBF.
|
||||||
|
- use_tiles_ignore_pbf=True
|
||||||
|
# Hard cap so a runaway tile build gets killed instead of the OOM killer
|
||||||
|
# picking off production containers on this 3.8 GB host.
|
||||||
|
mem_limit: 2g
|
||||||
|
volumes:
|
||||||
|
- ./valhalla_tiles:/custom_files
|
||||||
|
# Loopback-only: the API reaches Valhalla over the docker network, so this
|
||||||
|
# is purely for local probing. Valhalla has no auth — do not bind 0.0.0.0.
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8002:8002"
|
||||||
|
networks:
|
||||||
|
- frontend
|
||||||
|
|||||||
@@ -11,3 +11,4 @@ ortools
|
|||||||
python-dateutil
|
python-dateutil
|
||||||
faiss-cpu
|
faiss-cpu
|
||||||
psycopg2-binary
|
psycopg2-binary
|
||||||
|
redis
|
||||||
|
|||||||
Reference in New Issue
Block a user