/root/Routes-api on 31.97.228.132 is not a git repository. Work had been done directly on the box and existed nowhere else -- a single rm -rf from being lost, and impossible to review or roll back. Deploying the previous HEAD over it would have silently reverted all of this. Most visibly the Valhalla road-backend probe in main.py, whose own comment explains why it exists: road sequencing degrades to aerial silently by design, so an unreachable backend stays invisible, "which is exactly how the expired Google key went unnoticed". Overwriting it would have reintroduced precisely the failure it was written to catch, and the service would have kept answering 200 throughout. The server had also moved from Google Maps to Valhalla for road distance (VALHALLA_URL, road_backend_status, +190 lines in route_optimizer), extended docker-compose from 44 to 95 lines, and changed rider fetching, health, dynamic config and the cache layer. Only 10 files differ in substance. The other 27 that appeared to differ were CRLF-vs-LF noise -- the server writes CRLF -- and are normalised to LF here rather than committed as spurious whole-file rewrites. Committed as-is, before any change of mine, so the diff that follows is reviewable against what is actually running. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
118 lines
3.6 KiB
Python
118 lines
3.6 KiB
Python
"""Professional health check endpoints."""
|
|
|
|
import time
|
|
import logging
|
|
import sys
|
|
from typing import Optional
|
|
from datetime import datetime
|
|
from fastapi import APIRouter, Request
|
|
from pydantic import BaseModel, Field
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
router = APIRouter(prefix="/api/v1/health", tags=["Health"])
|
|
|
|
start_time = time.time()
|
|
|
|
|
|
class HealthResponse(BaseModel):
|
|
"""Health check response model."""
|
|
status: str = Field(..., description="Service status")
|
|
uptime_seconds: float = Field(..., description="Service uptime in seconds")
|
|
version: str = Field("2.0.0", description="API version")
|
|
timestamp: str = Field(..., description="Health check timestamp (ISO 8601)")
|
|
request_id: Optional[str] = Field(None, description="Request ID for tracing")
|
|
|
|
|
|
@router.get("/", response_model=HealthResponse)
|
|
async def health_check(request: Request):
|
|
"""
|
|
Health check endpoint.
|
|
|
|
Returns the current health status of the API service including:
|
|
- Service status (healthy/unhealthy)
|
|
- Uptime in seconds
|
|
- API version
|
|
- Timestamp
|
|
"""
|
|
try:
|
|
uptime = time.time() - start_time
|
|
request_id = getattr(request.state, "request_id", None)
|
|
|
|
return HealthResponse(
|
|
status="healthy",
|
|
uptime_seconds=round(uptime, 2),
|
|
version="2.0.0",
|
|
timestamp=datetime.utcnow().isoformat() + "Z",
|
|
request_id=request_id
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Health check failed: {e}", exc_info=True)
|
|
request_id = getattr(request.state, "request_id", None)
|
|
|
|
return HealthResponse(
|
|
status="unhealthy",
|
|
uptime_seconds=0.0,
|
|
version="2.0.0",
|
|
timestamp=datetime.utcnow().isoformat() + "Z",
|
|
request_id=request_id
|
|
)
|
|
|
|
|
|
@router.get("/ready")
|
|
async def readiness_check(request: Request):
|
|
"""
|
|
Readiness check endpoint for load balancers.
|
|
|
|
Returns 200 if the service is ready to accept requests.
|
|
"""
|
|
try:
|
|
# Check if critical services are available
|
|
# Add your service health checks here
|
|
|
|
return {
|
|
"status": "ready",
|
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
|
"request_id": getattr(request.state, "request_id", None)
|
|
}
|
|
except Exception as e:
|
|
logger.error(f"Readiness check failed: {e}")
|
|
return {
|
|
"status": "not_ready",
|
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
|
"request_id": getattr(request.state, "request_id", None)
|
|
}
|
|
|
|
|
|
@router.get("/road-backend")
|
|
async def road_backend_check(request: Request):
|
|
"""
|
|
Status of the Valhalla road-matrix backend used for road-aware stop sequencing.
|
|
|
|
Deliberately separate from /ready: an unreachable backend is a degradation,
|
|
not an outage — routing falls back to aerial ordering and keeps serving. This
|
|
endpoint is how you notice the degradation.
|
|
"""
|
|
from app.services.routing.route_optimizer import road_backend_status
|
|
|
|
status = await road_backend_status()
|
|
return {
|
|
**status,
|
|
"sequencing_mode": "road" if status.get("reachable") else "aerial",
|
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
|
"request_id": getattr(request.state, "request_id", None),
|
|
}
|
|
|
|
|
|
@router.get("/live")
|
|
async def liveness_check(request: Request):
|
|
"""
|
|
Liveness check endpoint for container orchestration.
|
|
|
|
Returns 200 if the service is alive.
|
|
"""
|
|
return {
|
|
"status": "alive",
|
|
"timestamp": datetime.utcnow().isoformat() + "Z",
|
|
"request_id": getattr(request.state, "request_id", None)
|
|
} |