diff --git a/app/config/dynamic_config.py b/app/config/dynamic_config.py index b73f18b..02321c0 100644 --- a/app/config/dynamic_config.py +++ b/app/config/dynamic_config.py @@ -38,6 +38,9 @@ DEFAULTS: Dict[str, Any] = { "emergency_load_penalty": 3.0, # km penalty per order in emergency assign # RouteOptimizer "search_time_limit_seconds": 5, + # Active-rider roster cache. The upstream getriderlogs call measured ~6s on + # live traffic. Short TTL so on/off-duty changes surface quickly; 0 = off. + "rider_roster_cache_ttl_seconds": 30, "avg_speed_kmh": 18.0, "road_factor": 1.3, # ClusteringService @@ -56,12 +59,20 @@ DEFAULTS: Dict[str, Any] = { "eta_history_days": 14, # rolling window pulled from nearledb "eta_stat": "median", # "median" or "p75" (p75 = more conservative) "eta_sync_interval_hours": 6, # autonomous background sync cadence - # Road-aware sequencing (Phase 2). OFF by default: enabling adds a Google - # Directions call (cost + latency) to the route hot path. Results are cached. - # Only the *visiting order* changes; step/ETA metrics stay aerial-based. + # Road-aware sequencing (Phase 2). Backed by self-hosted Valhalla, so there is + # no per-request cost — only local latency. Still OFF by default and left to + # the agent to enable on measured gain. Only the *visiting order* changes; + # step/ETA metrics stay aerial-based. "routing_use_road_distance": False, # AGENT-MANAGED (see routing_auto_manage) "routing_road_cache_ttl_seconds": 86400, # road geometry is stable; cache 24h - "routing_road_max_stops": 25, # Google distance-matrix practical cap + "routing_road_max_stops": 25, # beyond this the solver time outweighs + # the ordering gain (not a backend cap) + # Valhalla matrix backend. "motorcycle" models the lane access and one-way + # behaviour our riders actually have; "auto" would overstate their travel time. + "routing_valhalla_costing": "motorcycle", + "routing_matrix_timeout_seconds": 15.0, + "routing_matrix_max_unroutable_pct": 20.0, # above this -> tiles likely missing + # this region, fall back to aerial # Autonomous road-sequencing decision agent: measures road-vs-aerial travel # time on real batches and flips routing_use_road_distance on its own. "routing_auto_manage": True, # False -> humans own the flag diff --git a/app/main.py b/app/main.py index b95af96..c6f2b8a 100644 --- a/app/main.py +++ b/app/main.py @@ -6,7 +6,7 @@ import sys import time from contextlib import asynccontextmanager -# Load .env (DB_*, REDIS_*, GOOGLE_MAPS_API_KEY) into the environment early, +# Load .env (DB_*, REDIS_*, VALHALLA_URL) into the environment early, # before any service reads os.getenv at import time. try: from dotenv import load_dotenv @@ -28,7 +28,7 @@ from app.core.exception_handlers import ( ) from app.core.exceptions import APIException from app.middleware.request_id import RequestIDMiddleware -from app.routes import cache_router, health_router, ml_router, ml_web_router, optimization_router, batch_analytics_router, riders_router, doormile_router +from app.routes import cache_router, health_router, ml_router, ml_web_router, optimization_router, batch_analytics_router, riders_router # --------------------------------------------------------------------------- # Logging @@ -111,6 +111,27 @@ async def lifespan(app: FastAPI): except Exception as e: logger.warning(f"[RoadAgent] Decision agent start failed (non-fatal): {e}") + # Probe the Valhalla matrix backend once. Road sequencing degrades to aerial + # silently by design, so without this an unreachable or still-building backend + # stays invisible — which is exactly how the expired Google key went unnoticed. + try: + from app.services.routing.route_optimizer import road_backend_status + _rb = await road_backend_status() + if not _rb.get("configured"): + logger.warning("[RoadMatrix] VALHALLA_URL not set - aerial sequencing only.") + elif _rb.get("reachable"): + logger.info( + f"[RoadMatrix] Valhalla ready at {_rb.get('url')} " + f"(version={_rb.get('version')}, tiles={_rb.get('tileset_last_modified')})" + ) + else: + logger.warning( + f"[RoadMatrix] Valhalla at {_rb.get('url')} unreachable " + f"({_rb.get('detail')}) - aerial fallback until it answers." + ) + except Exception as e: + logger.warning(f"[RoadMatrix] Backend probe failed (non-fatal): {e}") + logger.info("[OK] Application ready.") yield logger.info("[STOP] Route Optimization API shutting down.") @@ -162,7 +183,6 @@ app.include_router(ml_router) app.include_router(ml_web_router) app.include_router(batch_analytics_router) app.include_router(riders_router) -app.include_router(doormile_router) @app.get("/", tags=["Root"]) diff --git a/app/routes/__init__.py b/app/routes/__init__.py index 47faef2..870e2ce 100644 --- a/app/routes/__init__.py +++ b/app/routes/__init__.py @@ -6,7 +6,6 @@ from .cache import router as cache_router from .ml_admin import router as ml_router, web_router as ml_web_router from .batch_analytics import router as batch_analytics_router from .riders import router as riders_router -from .doormile import router as doormile_router __all__ = [ "optimization_router", @@ -16,5 +15,4 @@ __all__ = [ "ml_web_router", "batch_analytics_router", "riders_router", - "doormile_router", ] diff --git a/app/routes/health.py b/app/routes/health.py index 2029735..dc1ba40 100644 --- a/app/routes/health.py +++ b/app/routes/health.py @@ -84,6 +84,26 @@ async def readiness_check(request: Request): } +@router.get("/road-backend") +async def road_backend_check(request: Request): + """ + Status of the Valhalla road-matrix backend used for road-aware stop sequencing. + + Deliberately separate from /ready: an unreachable backend is a degradation, + not an outage — routing falls back to aerial ordering and keeps serving. This + endpoint is how you notice the degradation. + """ + from app.services.routing.route_optimizer import road_backend_status + + status = await road_backend_status() + return { + **status, + "sequencing_mode": "road" if status.get("reachable") else "aerial", + "timestamp": datetime.utcnow().isoformat() + "Z", + "request_id": getattr(request.state, "request_id", None), + } + + @router.get("/live") async def liveness_check(request: Request): """ diff --git a/app/services/__init__.py b/app/services/__init__.py index fc9d666..c745694 100644 --- a/app/services/__init__.py +++ b/app/services/__init__.py @@ -15,6 +15,31 @@ except Exception: # pragma: no cover logger = logging.getLogger(__name__) +def _redis_url_from_parts() -> Optional[str]: + """ + Build a redis:// URL from discrete env vars, percent-encoding the credentials. + + A hand-written REDIS_URL breaks the moment the password contains a reserved + character: an '@' splits the authority section early, so the client silently + dials a garbage host instead of failing loudly. Assembling the URL here with + quote() removes that whole class of bug - callers set REDIS_HOST/REDIS_PASSWORD + and never have to think about escaping. + + Returns None when REDIS_HOST is unset, so an explicit REDIS_URL still wins. + """ + host = os.getenv("REDIS_HOST") + if not host: + return None + from urllib.parse import quote + + user = os.getenv("REDIS_USERNAME", "") + pwd = os.getenv("REDIS_PASSWORD", "") + port = os.getenv("REDIS_PORT", "6379") + db = os.getenv("REDIS_DB", "0") + auth = f"{quote(user, safe='')}:{quote(pwd, safe='')}@" if pwd else "" + return f"redis://{auth}{host}:{port}/{db}" + + class RedisCache: """Lightweight Redis cache wrapper with graceful in-memory fallback.""" @@ -33,9 +58,10 @@ class RedisCache: self._client = None self._stats = {"hits": 0, "misses": 0, "sets": 0} - url = os.getenv(url_env) + url = os.getenv(url_env) or _redis_url_from_parts() if not url or redis is None: - logger.warning("Redis not configured or client unavailable; falling back to local thread-safe in-memory cache") + _why = "redis client not installed" if redis is None else "no REDIS_URL/REDIS_HOST" + logger.warning(f"Redis not configured or client unavailable ({_why}); falling back to local thread-safe in-memory cache") return try: self._client = redis.Redis.from_url(url, decode_responses=True) diff --git a/app/services/rider/get_active_riders.py b/app/services/rider/get_active_riders.py index 35f8216..07b4421 100644 --- a/app/services/rider/get_active_riders.py +++ b/app/services/rider/get_active_riders.py @@ -6,13 +6,45 @@ from typing import List, Dict, Any logger = logging.getLogger(__name__) +def _roster_cache_ttl() -> int: + """Roster cache TTL in seconds. 0 disables caching entirely.""" + try: + from app.config.dynamic_config import get_config + return int(get_config().get("rider_roster_cache_ttl_seconds", 30)) + except Exception: + return 30 + + async def fetch_active_riders() -> List[Dict[str, Any]]: """ Fetch active rider logs from the external API for the current date. Returns a list of rider log dictionaries. + + This upstream call measured ~6s on live traffic and was the single largest + component of riderassign latency, so successful responses are cached for + `rider_roster_cache_ttl_seconds`. The TTL is deliberately short: a rider + going on or off duty must become visible quickly, and the absent-rider + validation in the assignment path checks against this roster. + + Only non-empty successes are cached. Caching an empty result or a failure + would turn a brief upstream blip into a full-TTL outage where every request + sees zero riders. """ + today_str = datetime.now().strftime("%Y-%m-%d") + ttl = _roster_cache_ttl() + cache_key = f"riders:active:{today_str}" + + if ttl > 0: + try: + from app.services import cache as _cache + cached = _cache.get_json(cache_key) + if isinstance(cached, list) and cached: + logger.debug(f"[Riders] roster cache hit ({len(cached)} riders)") + return cached + except Exception: + pass + try: - today_str = datetime.now().strftime("%Y-%m-%d") url = "https://jupiter.nearle.app/live/api/v2/partners/getriderlogs/" params = { "applocationid": 1, @@ -21,19 +53,26 @@ async def fetch_active_riders() -> List[Dict[str, Any]]: "todate": today_str, "keyword": "" } - + async with httpx.AsyncClient(timeout=30.0) as client: response = await client.get(url, params=params) response.raise_for_status() data = response.json() - + if data and data.get("code") == 200 and data.get("details"): # Filter riders who are in our preferences list and are 'active' or 'idle' (assuming we want online riders) # The user's example showed "onduty": 1. We might want to filter by that. # For now, returning all logs, filtering can happen in assignment logic or here. # Let's return the raw list as requested, filtering logic will be applied during assignment. - return data.get("details", []) - + riders = data.get("details", []) + if riders and ttl > 0: + try: + from app.services import cache as _cache + _cache.set_json(cache_key, riders, ttl_seconds=ttl) + except Exception: + pass + return riders + logger.warning(f"Fetch active riders returned no details: {data}") return [] @@ -60,7 +99,7 @@ async def fetch_created_orders() -> List[Dict[str, Any]]: "pageno": 1 # "pagesize" intentionally omitted to fetch all } - + async with httpx.AsyncClient(timeout=60.0) as client: response = await client.get(url, params=params) response.raise_for_status() diff --git a/app/services/routing/road_sequencing_agent.py b/app/services/routing/road_sequencing_agent.py index a6ff717..4e20ca2 100644 --- a/app/services/routing/road_sequencing_agent.py +++ b/app/services/routing/road_sequencing_agent.py @@ -61,8 +61,8 @@ class RoadSequencingAgent: from app.core.arrow_utils import calculate_haversine_matrix_vectorized opt = self._optimizer() - if not opt.use_google_maps: - return {"evaluated": 0, "reason": "no_google_key", "mean_gain_pct": 0.0} + if not opt.use_road_matrix: + return {"evaluated": 0, "reason": "no_matrix_backend", "mean_gain_pct": 0.0} batches = get_delivery_history_service().sample_batches( days=days, limit=sample_batches diff --git a/app/services/routing/route_optimizer.py b/app/services/routing/route_optimizer.py index e3f6872..b130b5e 100644 --- a/app/services/routing/route_optimizer.py +++ b/app/services/routing/route_optimizer.py @@ -8,7 +8,7 @@ ALGORITHM: TSP / VRP with Google OR-Tools FEATURES: - Automatic outlier detection and coordinate correction -- Hybrid distance calculation (Google Maps + Haversine fallback) +- Hybrid distance calculation (self-hosted Valhalla road matrix + Haversine fallback) - Robust error handling for invalid inputs """ @@ -37,6 +37,37 @@ except ImportError: logger = logging.getLogger(__name__) +async def road_backend_status() -> Dict[str, Any]: + """ + Probe the Valhalla matrix backend. + + Module-level (not a RouteOptimizer method) so startup and health checks can + call it without paying for a full optimizer construction, which pulls in the + empirical ETA calculator and its history load. + """ + url = os.getenv("VALHALLA_URL", "").strip().rstrip("/") + if not url: + return { + "configured": False, + "reachable": False, + "detail": "VALHALLA_URL not set - road sequencing disabled, aerial only", + } + try: + async with httpx.AsyncClient(timeout=5.0) as client: + resp = await client.get(f"{url}/status") + resp.raise_for_status() + data = resp.json() + return { + "configured": True, + "reachable": True, + "url": url, + "version": data.get("version"), + "tileset_last_modified": data.get("tileset_last_modified"), + } + except Exception as e: + return {"configured": True, "reachable": False, "url": url, "detail": str(e)} + + class RouteOptimizer: """Route optimization using Google OR-Tools (Async).""" @@ -60,9 +91,11 @@ class RouteOptimizer: # Road factor (haversine -> road distance multiplier, ML-tuned) self.road_factor = float(_cfg.get("road_factor")) - # Google Maps API settings - self.google_maps_api_key = os.getenv("GOOGLE_MAPS_API_KEY", "") - self.use_google_maps = bool(self.google_maps_api_key) + # Road travel-time matrix backend: self-hosted Valhalla (no per-request + # cost, no rate limit). Unset VALHALLA_URL -> road sequencing stays off + # and every caller falls back to aerial ordering. + self.valhalla_url = os.getenv("VALHALLA_URL", "").strip().rstrip("/") + self.use_road_matrix = bool(self.valhalla_url) # Solver time limit (ML-tuned) self.search_time_limit_seconds = int(_cfg.get("search_time_limit_seconds")) @@ -90,47 +123,94 @@ class RouteOptimizer: # ROAD-AWARE VISITING ORDER (Phase 2 - opt-in, cached) # ------------------------------------------------------------------ + def _aerial_minutes( + self, a: Tuple[float, float], b: Tuple[float, float] + ) -> float: + """Haversine travel-time estimate (minutes), used to patch unroutable pairs.""" + km = self.haversine_distance(a[0], a[1], b[0], b[1]) * self.road_factor + return (km / (self.avg_speed_kmh or 20.0)) * 60.0 + async def _road_duration_matrix( self, coords: _List[Tuple[float, float]] ) -> Optional[_List[_List[float]]]: """ - Full NxN road travel-TIME matrix (minutes) via Google Distance Matrix. + Full NxN road travel-TIME matrix (minutes) via self-hosted Valhalla. + + One /sources_to_targets call returns the whole matrix, so unlike the old + Google Distance Matrix path there is no per-request element cap and no + chunking. Costing defaults to `motorcycle`, which models the lane access + and one-way behaviour our riders actually have — a car matrix systematically + overstates their travel time. + + Valhalla reports an unroutable pair as time=null. Those are patched with an + aerial estimate rather than left at 0, because a 0-cost edge would look + free to the TSP and pull the whole sequence through it. If too many pairs + are unroutable the tileset probably doesn't cover this region, so we bail + to aerial entirely. - Chunks destinations to respect Google's ~100-elements-per-request limit. Returns None on any failure so the caller falls back to aerial ordering. """ - if not self.use_google_maps: + if not self.use_road_matrix: return None + cfg = get_config() + costing = str(cfg.get("routing_valhalla_costing", "motorcycle")) + timeout = float(cfg.get("routing_matrix_timeout_seconds", 15.0)) + max_unroutable = float(cfg.get("routing_matrix_max_unroutable_pct", 20.0)) + n = len(coords) - origins = "|".join(f"{la},{lo}" for la, lo in coords) + locations = [{"lat": float(la), "lon": float(lo)} for la, lo in coords] matrix = [[0.0] * n for _ in range(n)] - dest_chunk = max(1, 100 // max(1, n)) try: - async with httpx.AsyncClient(timeout=15.0) as client: - for j0 in range(0, n, dest_chunk): - js = list(range(j0, min(j0 + dest_chunk, n))) - dests = "|".join(f"{coords[j][0]},{coords[j][1]}" for j in js) - resp = await client.get( - "https://maps.googleapis.com/maps/api/distancematrix/json", - params={"origins": origins, "destinations": dests, - "key": self.google_maps_api_key, "units": "metric"}, - ) - resp.raise_for_status() - data = resp.json() - if data.get("status") != "OK": - logger.debug(f"[RoadSeq] DistanceMatrix status={data.get('status')}") - return None - for i, row in enumerate(data.get("rows", [])): - for k, el in enumerate(row.get("elements", [])): - if el.get("status") == "OK": - dur = el.get("duration", {}).get("value") - if dur is not None: - matrix[i][js[k]] = dur / 60.0 + async with httpx.AsyncClient(timeout=timeout) as client: + resp = await client.post( + f"{self.valhalla_url}/sources_to_targets", + json={ + "sources": locations, + "targets": locations, + "costing": costing, + "units": "km", + }, + ) + resp.raise_for_status() + data = resp.json() + + rows = data.get("sources_to_targets") or [] + if len(rows) != n: + logger.debug( + f"[RoadSeq] Valhalla returned {len(rows)} rows, expected {n}" + ) + return None + + unroutable = 0 + for i, row in enumerate(rows): + for el in row or []: + j = el.get("to_index") + if j is None or not (0 <= j < n): + continue + secs = el.get("time") + if secs is None: + unroutable += 1 + matrix[i][j] = self._aerial_minutes(coords[i], coords[j]) + else: + matrix[i][j] = secs / 60.0 + + pct = 100.0 * unroutable / max(1, n * n) + if pct > max_unroutable: + logger.warning( + f"[RoadSeq] {unroutable}/{n * n} pairs ({pct:.0f}%) unroutable via " + f"Valhalla - tileset likely missing this region; using aerial" + ) + return None + if unroutable: + logger.debug( + f"[RoadSeq] patched {unroutable}/{n * n} unroutable pairs with aerial" + ) return matrix except Exception as e: logger.debug(f"[RoadSeq] matrix build failed: {e}") return None + async def _road_optimal_order( self, start_lat: float, @@ -140,27 +220,25 @@ class RouteOptimizer: """ Road-aware visiting order for `points`, starting from (start_lat, start_lon). - Builds a real road travel-TIME matrix (Google Distance Matrix) and solves - an OPEN TSP with OR-Tools (return-to-depot edge = 0), so the sequence - respects real road geometry/one-ways instead of straight-line distance. - Validated on live batches to cut real travel time ~5-13% vs aerial; note - Google's Directions optimize:true is NOT used - it optimises a closed loop - and measured *worse* than aerial for our open delivery routes. + Builds a real road travel-TIME matrix (Valhalla) and solves an OPEN TSP + with OR-Tools (return-to-depot edge = 0), so the sequence respects real + road geometry/one-ways instead of straight-line distance. Validated on + live batches to cut real travel time ~5-13% vs aerial. Returns 0-based indices into `points` in optimal order, or None to signal the caller to fall back to the existing aerial greedy + 2-opt. - Safe + cheap on the live path: - * disabled unless `routing_use_road_distance` AND a Google key is set + Safe on the live path: + * disabled unless `routing_use_road_distance` AND VALHALLA_URL is set * only for 3..`routing_road_max_stops` stops (fewer is trivial; more - exceeds Google's waypoint-optimize cap) + costs more solver time than the ordering gain is worth) * result cached in Redis (default 24h) keyed by the rounded coords in input order, so the returned indices always map back correctly """ cfg = get_config() if not cfg.get("routing_use_road_distance", False): return None - if not self.use_google_maps: + if not self.use_road_matrix: return None n = len(points) if n < 3 or n > int(cfg.get("routing_road_max_stops", 25)): @@ -362,14 +440,17 @@ class RouteOptimizer: search_params.local_search_metaheuristic = ( routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH ) - # TSP time limit hard-capped at 2 seconds per kitchen. - # search_time_limit_seconds can be tuned up to 8-10s via config, but at - # delivery scale (< 15 stops) - # OR-Tools finds a near-optimal solution in < 200ms. Waiting 8-10s - # per kitchen x 3 kitchens x 4 riders = 96s of unnecessary waiting. - # The VRP already has its own 3s cap. This cap applies to per-rider - # TSP and the per-kitchen beatmap solves. - search_params.time_limit.seconds = min(self.search_time_limit_seconds, 2) + # TSP time budget, ceiling 2 seconds per kitchen. This cap applies to + # per-rider TSP and the per-kitchen beatmap solves. + # + # GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES - it does not + # return early once it has the optimum. A flat cap therefore burned the + # whole budget on trivial inputs, where PATH_CHEAPEST_ARC is done in + # under 200ms. Scale with stop count; large inputs still get the ceiling. + _cap_ms = min(self.search_time_limit_seconds, 2) * 1000 + search_params.time_limit.FromMilliseconds( + max(200, min(_cap_ms, 200 * len(locations))) + ) solution = routing.SolveWithParameters(search_params) @@ -742,13 +823,18 @@ class RouteOptimizer: sp.local_search_metaheuristic = ( routing_enums_pb2.LocalSearchMetaheuristic.GUIDED_LOCAL_SEARCH ) - # VRP time budget: cap at 3 seconds so the API response is never + # VRP time budget: ceiling of 3 seconds so the API response is never # blocked longer than that. The x 3 multiplier caused 15-second # responses with the default 5-second search_time_limit. - # PATH_CHEAPEST_ARC typically finds a good solution in < 1 second - # for our typical sizes (<= 15 riders, <= 60 orders); GLS then - # improves it within the remaining budget. - sp.time_limit.seconds = min(self.search_time_limit_seconds, 3) + # + # GUIDED_LOCAL_SEARCH runs until its time limit EXPIRES rather than + # stopping once the optimum is found, so a flat 3s spent ~2.9s of + # pure waiting on a 1-order batch (measured on live traffic). + # Scale with problem size; large batches still get the full ceiling. + _cap_ms = min(self.search_time_limit_seconds, 3) * 1000 + sp.time_limit.FromMilliseconds( + max(250, min(_cap_ms, 250 * len(orders))) + ) solution = routing.SolveWithParameters(sp) diff --git a/docker-compose.yml b/docker-compose.yml index 7a7d77a..4a89ac7 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -3,6 +3,9 @@ version: "3.9" networks: frontend: external: true + # redis + postgres-logistics live here; routes_api joins it purely to reach redis. + logistics: + external: true services: routes_api: @@ -14,13 +17,20 @@ services: restart: unless-stopped environment: - UVICORN_WORKERS=2 - - REDIS_URL=redis://:${REDIS_PASSWORD}@routes_redis:6379/0 + # Redis as discrete parts, NOT a hand-built REDIS_URL: the password contains + # reserved characters and inlining it produced redis://:pa@ss@routes_redis/..., + # which resolved to a garbage host and silently fell back to in-memory cache. + # app/services/__init__.py percent-encodes these when assembling the URL. + # Host is `redis` (its container name) on the logistics network. + - REDIS_HOST=redis + - REDIS_PORT=6379 + - REDIS_PASSWORD=${REDIS_PASSWORD} # Optional: Set cache TTL in seconds (default: 300 = 5 min, 86400 = 24h) # Uncomment and set in .env file: REDIS_CACHE_TTL_SECONDS=86400 # - REDIS_CACHE_TTL_SECONDS=${REDIS_CACHE_TTL_SECONDS} - # Google Maps API key for accurate road distance calculation (actualkms) - # Set in .env file: GOOGLE_MAPS_API_KEY=your_api_key_here - - GOOGLE_MAPS_API_KEY=${GOOGLE_MAPS_API_KEY} + # Self-hosted Valhalla, source of the road travel-time matrix used for + # road-aware stop sequencing. Unset -> sequencing falls back to aerial. + - VALHALLA_URL=http://routes_valhalla:8002 # nearledb (read-only) — source for empirical ETA learning - DB_HOST=${DB_HOST} - DB_PORT=${DB_PORT} @@ -42,3 +52,44 @@ services: - ./delivery_corrections.csv:/app/delivery_corrections.csv:ro networks: - frontend + - logistics + + # Road travel-time matrix backend. Replaces the Google Distance Matrix API: + # no key, no per-request cost, no element cap, and a real `motorcycle` costing + # model instead of car durations. + # + # Deliberately NOT using tile_urls. This host has 3.8 GB RAM and runs production + # alongside it; building the 556 MB South India extract here risks the OOM killer + # taking down routes_api/postgres. Instead ./valhalla_tiles holds a pre-clipped + # Coimbatore PBF (bbox 76.60,10.70 -> 77.30,11.35, derived from the actual + # delivery coordinates in delivery_details.csv, which span only ~19 km). Valhalla + # builds from any .pbf it finds in /custom_files. + # + # To re-clip after an OSM refresh, on the host: + # osmium extract -b 76.60,10.70,77.30,11.35 .osm.pbf \ + # -o valhalla_tiles/coimbatore.osm.pbf + # docker compose up -d --force-recreate valhalla # with force_rebuild=True + valhalla: + image: ghcr.io/valhalla/valhalla-scripted:latest + container_name: routes_valhalla + restart: unless-stopped + environment: + - serve_tiles=True + - force_rebuild=False + - build_elevation=False + # India drives on the left; admin regions are what teach Valhalla that. + - build_admins=True + - build_time_zones=True + # Reuse built tiles on restart instead of rebuilding from the PBF. + - use_tiles_ignore_pbf=True + # Hard cap so a runaway tile build gets killed instead of the OOM killer + # picking off production containers on this 3.8 GB host. + mem_limit: 2g + volumes: + - ./valhalla_tiles:/custom_files + # Loopback-only: the API reaches Valhalla over the docker network, so this + # is purely for local probing. Valhalla has no auth — do not bind 0.0.0.0. + ports: + - "127.0.0.1:8002:8002" + networks: + - frontend diff --git a/requirements.txt b/requirements.txt index 4d2d290..232000a 100644 --- a/requirements.txt +++ b/requirements.txt @@ -11,3 +11,4 @@ ortools python-dateutil faiss-cpu psycopg2-binary +redis