updates on the admincontroller and the hubcity fix and queue orders

This commit is contained in:
2026-09-25 16:30:38 +05:30
parent ee79c80338
commit 44ba33eda2
7 changed files with 441 additions and 31 deletions

View File

@@ -2,6 +2,9 @@ package assignment
import (
"encoding/json"
"os"
"strconv"
"strings"
"time"
"doormile/db"
@@ -43,6 +46,102 @@ const (
type assignmentRequest struct {
BookingID int `json:"booking_id"`
Kind string `json:"kind"`
// Extended retry. One "round" is one JetStream message with up to
// maxRetries deliveries (~8 minutes). A round that ends with no miler
// found re-queues the booking as a new round, until the retry window
// runs out. All three are omitempty so messages already on the stream
// (published before this field existed) still decode: they are treated
// as round 1 and their window starts when first seen.
FirstQueuedAt int64 `json:"first_queued_at,omitempty"` // unix seconds, first enqueue
Round int `json:"round,omitempty"` // 1-based
NotBefore int64 `json:"not_before,omitempty"` // unix seconds; wait until then before attempting
}
// ---- Extended retry --------------------------------------------------------
//
// Auto-assignment used to give up for good after one round — 5 attempts over
// ~8 minutes. A booking created while every nearby rider was full, on a
// break, or not yet broadcasting GPS then sat in pending_pickup forever, even
// after riders freed up minutes later: nothing ever looked at it again, and
// nothing on the booking said so. That is how DM-664517 got stuck.
//
// Now a failed round re-queues the booking and keeps trying every
// extendedRetryDelay until the retry window closes (default 2h, env
// ASSIGNMENT_RETRY_WINDOW_MINUTES). Each attempt re-reads the booking and
// stops as soon as it is cancelled or has a rider — including one assigned by
// hand — so a manual assignment ends the loop.
//
// maxRetries is deliberately NOT raised to get this. It is also the durable
// consumer's MaxDeliver, which is stored on the NATS server; subscribing with
// a different value fails ("subscribe failed") and the worker would then stop
// assigning anything at all. Re-queuing new rounds keeps the consumer config
// identical.
//
// booking.assignment_failed still fires once, at the end of the FIRST round,
// exactly when it did before — the DispatchAgent and ops alerting keep their
// timing and aren't sent a duplicate for every later round.
const (
extendedRetryDelay = 5 * time.Minute
defaultRetryWindowMinutes = 120
)
// retryWindow reads ASSIGNMENT_RETRY_WINDOW_MINUTES per call, like
// maxActiveBookings, so it can be tuned without a redeploy. 0 restores the old
// single-round behaviour; a non-numeric or negative value falls back to the
// default rather than disabling retries by typo.
func retryWindow() time.Duration {
if v := strings.TrimSpace(os.Getenv("ASSIGNMENT_RETRY_WINDOW_MINUTES")); v != "" {
if n, err := strconv.Atoi(v); err == nil && n >= 0 {
return time.Duration(n) * time.Minute
}
utils.Warn("ASSIGNMENT_RETRY_WINDOW_MINUTES is not a non-negative integer, using the default",
"value", v, "default_minutes", defaultRetryWindowMinutes)
}
return defaultRetryWindowMinutes * time.Minute
}
// withinRetryWindow reports whether another round may start now.
func withinRetryWindow(firstQueuedAt int64, now time.Time) bool {
if firstQueuedAt == 0 {
return true
}
return now.Sub(time.Unix(firstQueuedAt, 0)) < retryWindow()
}
// requeueNextRound publishes the booking as a new round that waits
// extendedRetryDelay before its first attempt. Returns false when it could not
// be queued (JetStream down or publish error) — the caller then gives up as
// the old code did, rather than dropping it silently.
func requeueNextRound(req assignmentRequest, now time.Time) bool {
if db.Js == nil {
return false
}
next := req
if next.FirstQueuedAt == 0 {
next.FirstQueuedAt = now.Unix()
}
if next.Round < 1 {
next.Round = 1
}
next.Round++
next.NotBefore = now.Add(extendedRetryDelay).Unix()
data, err := json.Marshal(next)
if err != nil {
utils.Error("Assignment: marshal next round failed", "booking_id", req.BookingID, "error", err)
return false
}
if _, err := db.Js.Publish(subjectAssignmentRequested, data); err != nil {
utils.Error("Assignment: publish next round failed", "booking_id", req.BookingID, "error", err)
return false
}
utils.Warn("Assignment: no miler this round, retrying later",
"booking_id", req.BookingID,
"next_round", next.Round,
"retry_in", extendedRetryDelay.String(),
)
return true
}
// enqueue publishes an assignment request, falling back to the old in-process
@@ -60,7 +159,12 @@ func enqueue(bookingID int, kind string) {
return
}
data, err := json.Marshal(assignmentRequest{BookingID: bookingID, Kind: kind})
data, err := json.Marshal(assignmentRequest{
BookingID: bookingID,
Kind: kind,
FirstQueuedAt: time.Now().Unix(),
Round: 1,
})
if err != nil {
utils.Error("Assignment: marshal failed, retrying in-process",
"booking_id", bookingID, "error", err)
@@ -149,6 +253,16 @@ func handleAssignmentMessage(msg *nats.Msg) {
return
}
// A later round waits before its first attempt. JetStream has no delayed
// publish, so the wait is a NAK — it uses one of the round's deliveries,
// leaving maxRetries-1 attempts, which is fine at this cadence.
if req.NotBefore > 0 {
if wait := time.Until(time.Unix(req.NotBefore, 0)); wait > time.Second {
_ = msg.NakWithDelay(wait)
return
}
}
// NumDelivered counts this delivery, so it runs 1..maxRetries.
attempt := 1
if md, err := msg.Metadata(); err == nil {
@@ -167,52 +281,99 @@ func handleAssignmentMessage(msg *nats.Msg) {
return
}
// Last delivery: JetStream will not redeliver past MaxDeliver, so the
// terminal failure has to be published here or it never fires at all. Ack
// rather than Nak so the message is not left to expire silently.
// Last delivery of this round: JetStream will not redeliver past
// MaxDeliver, so what happens next is decided here. Ack rather than Nak so
// the message is not left to expire silently.
if attempt >= maxRetries {
utils.Error("AssignmentWorker: NO_MILER_AVAILABLE — all attempts exhausted",
"booking_id", req.BookingID, "attempts", attempt)
publishAssignmentFailed(req.BookingID, reasonNoMilerAvailable)
round := req.Round
if round < 1 {
round = 1
}
if round == 1 {
utils.Error("AssignmentWorker: NO_MILER_AVAILABLE — first round exhausted",
"booking_id", req.BookingID, "attempts", attempt)
publishAssignmentFailed(req.BookingID, reasonNoMilerAvailable)
}
// Only "no miler found" earns another round. A hard error (booking
// gone, no pickup coordinates) won't fix itself by waiting.
now := time.Now()
if err == nil && withinRetryWindow(req.FirstQueuedAt, now) && requeueNextRound(req, now) {
_ = msg.Ack()
return
}
utils.Error("AssignmentWorker: NO_MILER_AVAILABLE — giving up",
"booking_id", req.BookingID,
"rounds", round,
"retry_window", retryWindow().String(),
"last_error", err,
)
_ = msg.Ack()
return
}
delay := retryDelay
if req.Round > 1 {
delay = extendedRetryDelay
}
utils.Warn("AssignmentWorker: no eligible miler, will retry",
"booking_id", req.BookingID,
"round", req.Round,
"attempt", attempt,
"remaining", maxRetries-attempt,
"retry_in", retryDelay.String(),
"retry_in", delay.String(),
)
_ = msg.NakWithDelay(retryDelay)
_ = msg.NakWithDelay(delay)
}
// runInline is the pre-JetStream behaviour, kept only as the fallback path when
// the event bus is down. It holds its retries in memory and does not survive a
// restart — which is exactly the weakness the queue exists to fix.
func runInline(bookingID int, kind string) {
for attempt := 1; attempt <= maxRetries; attempt++ {
firstQueuedAt := time.Now().Unix()
failurePublished := false
for attempt := 1; ; attempt++ {
if attempt > 1 {
time.Sleep(retryDelay)
if attempt <= maxRetries {
time.Sleep(retryDelay)
} else {
time.Sleep(extendedRetryDelay)
}
}
utils.Info("Assignment(inline): attempting", "booking_id", bookingID, "attempt", attempt)
done, err := attemptOnce(bookingID, kind)
if err != nil {
// Same rule as the queue: a hard error doesn't earn the extended
// window, but it still gets the original first-round attempts.
utils.Error("Assignment(inline): attempt error",
"booking_id", bookingID, "attempt", attempt, "error", err)
if attempt >= maxRetries {
break
}
continue
}
if done {
return
}
utils.Warn("Assignment(inline): no eligible miler",
"booking_id", bookingID, "attempt", attempt, "remaining", maxRetries-attempt)
if attempt == maxRetries && !failurePublished {
utils.Error("Assignment(inline): NO_MILER_AVAILABLE — first round exhausted",
"booking_id", bookingID, "attempts", attempt)
publishAssignmentFailed(bookingID, reasonNoMilerAvailable)
failurePublished = true
}
if attempt >= maxRetries && !withinRetryWindow(firstQueuedAt, time.Now()) {
break
}
utils.Warn("Assignment(inline): no eligible miler", "booking_id", bookingID, "attempt", attempt)
}
utils.Error("Assignment(inline): NO_MILER_AVAILABLE — all retries exhausted",
"booking_id", bookingID, "max_retries", maxRetries)
publishAssignmentFailed(bookingID, reasonNoMilerAvailable)
utils.Error("Assignment(inline): NO_MILER_AVAILABLE — giving up",
"booking_id", bookingID, "retry_window", retryWindow().String())
if !failurePublished {
publishAssignmentFailed(bookingID, reasonNoMilerAvailable)
}
}

View File

@@ -0,0 +1,78 @@
package assignment
import (
"encoding/json"
"testing"
"time"
)
func TestRetryWindowDefaultAndOverride(t *testing.T) {
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", "")
if got := retryWindow(); got != 120*time.Minute {
t.Fatalf("default window = %v, want 2h", got)
}
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", "45")
if got := retryWindow(); got != 45*time.Minute {
t.Fatalf("override window = %v, want 45m", got)
}
// 0 restores the old single-round behaviour.
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", "0")
if got := retryWindow(); got != 0 {
t.Fatalf("zero window = %v, want 0", got)
}
// A typo must not disable retries.
for _, bad := range []string{"abc", "-5", "1.5"} {
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", bad)
if got := retryWindow(); got != 120*time.Minute {
t.Fatalf("bad value %q gave %v, want the 2h default", bad, got)
}
}
}
func TestWithinRetryWindow(t *testing.T) {
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", "120")
now := time.Now()
if !withinRetryWindow(now.Add(-10*time.Minute).Unix(), now) {
t.Fatal("10 minutes in should still retry")
}
if withinRetryWindow(now.Add(-121*time.Minute).Unix(), now) {
t.Fatal("121 minutes in should give up")
}
// Messages queued before this change carry no timestamp; they get a
// window starting now rather than being dropped.
if !withinRetryWindow(0, now) {
t.Fatal("a message without first_queued_at should retry")
}
t.Setenv("ASSIGNMENT_RETRY_WINDOW_MINUTES", "0")
if withinRetryWindow(now.Add(-time.Second).Unix(), now) {
t.Fatal("window 0 should never start another round")
}
}
// Messages already on the ASSIGNMENTS stream when this ships were published
// with only booking_id and kind. They must still decode and act as round 1.
func TestAssignmentRequestDecodesOldPayload(t *testing.T) {
var req assignmentRequest
if err := json.Unmarshal([]byte(`{"booking_id":664517,"kind":"express"}`), &req); err != nil {
t.Fatalf("old payload failed to decode: %v", err)
}
if req.BookingID != 664517 || req.Kind != kindExpress {
t.Fatalf("decoded %+v", req)
}
if req.Round != 0 || req.FirstQueuedAt != 0 || req.NotBefore != 0 {
t.Fatalf("new fields should be zero on an old payload, got %+v", req)
}
}
// requeueNextRound must refuse (return false) rather than panic when
// JetStream is not connected, so the caller falls back to giving up cleanly.
func TestRequeueWithoutJetStream(t *testing.T) {
if requeueNextRound(assignmentRequest{BookingID: 1, Kind: kindExpress, Round: 1}, time.Now()) {
t.Fatal("requeue should report false with no JetStream connection")
}
}