Files
backend_fiesta/messaging/posmqtt.go
Suriya b9f389fcdf Process the MQTT ingest on a bounded worker pool
paho delivers on one goroutine, so bills were committed strictly one
after another. Each is a full Postgres transaction — advisory lock,
dedup, stock row locks, availability check, four inserts, commit — which
is 10-30ms, so the ceiling was roughly 30-100 bills a second and a
shop's backlog draining after an outage took minutes to land.

A fixed pool behind a bounded queue, rather than a goroutine per
message. Unbounded concurrency would open a transaction per message and
exhaust the connection pool under a storm, stalling every one of them at
once — a slow minute turned into a dead one. When the queue fills,
submit blocks: paho stops acknowledging, the broker's in-flight window
fills, it stops sending, and the backpressure reaches the till, which
holds its bills and retries. Slow, but nothing is dropped.

Heartbeats get their own pool. Sharing one would let a backlog of bills
delay presence, so every till would appear to go dark at exactly the
moment the system was busiest — the worst time to be blind to which
counters are alive.

Payloads are copied on the way in. paho reuses its buffer once a handler
returns and the work now happens after that, so a queued bill would
otherwise be read as whatever message arrived next — silently, and as
valid JSON often enough to commit the wrong sale.

One bug found by its own test: submit-after-stop selected between a
done-channel and the job channel, and once both were ready Go picks at
random. Picking the send panics on a closed channel. It would have shown
up in production as an occasional crash during shutdown and nowhere
else. Now guarded by an RWMutex held across the send, so the queue
cannot be closed under one in progress.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-03 19:53:39 +05:30

341 lines
12 KiB
Go

package messaging
import (
"context"
"encoding/json"
"fmt"
"log"
"os"
"strconv"
"strings"
"time"
"nearle/models"
"nearle/services"
mqtt "github.com/eclipse/paho.mqtt.golang"
)
// MQTT ingest for the Nearle POS terminals.
//
// The broker is Eclipse Mosquitto, shared with the rider fleet. An audit of the
// estate found no reachable NATS and no MQTT gateway on the NATS boxes that do
// exist, so a NATS consumer that briefly lived here was deleted rather than
// left to rot — a client for a protocol nothing speaks is worse than none.
//
// Enabled with MQTT_URL. Unset, the terminals reach the same service over HTTP
// instead, and this file does nothing.
//
// Only one replica consumes: see posConsumerElected.
const (
// Namespaced under `nearle/` alongside the rider app's
// `nearle/riders/{riderId}/...`, so one broker ACL rule covers each system
// and it is obvious from a topic which one it belongs to.
//
// Wildcards for MQTT are `+` per level, where NATS uses `*`.
topicOrders = "nearle/pos/+/+/order"
topicCustomers = "nearle/pos/+/+/customer"
topicHealth = "nearle/pos/+/+/health"
)
type PosMqttConsumer struct {
client mqtt.Client
svc services.PosService
// Bills and registrations share a pool; heartbeats get their own, so a
// backlog of sales cannot make every till look dark at the moment the
// system is busiest.
ingest *posPool
health *posPool
}
// StartPosMqttConsumer connects and subscribes.
//
// Returns (nil, nil) when MQTT_URL is unset — a deployment without a broker is
// supported, and the caller carries on with the HTTP endpoints.
func StartPosMqttConsumer(svc services.PosService) (*PosMqttConsumer, error) {
url := strings.TrimSpace(os.Getenv("MQTT_URL"))
if url == "" {
log.Println("pos: MQTT_URL not set, plain-MQTT ingest disabled")
return nil, nil
}
// Only one replica consumes.
//
// MQTT has no queue groups — every subscriber receives every message, so
// three replicas would each commit the same bill and publish three acks.
// The ingest is idempotent, so nothing double-counts, but it is three times
// the database work and three times the traffic for one sale.
//
// A StatefulSet gives pods stable ordinal names, so ordinal 0 is a
// deterministic election with no coordination and no extra dependency. If
// that pod dies the set recreates it; tills hold their bills and re-send in
// the meantime, which is exactly what they are built to do.
if !posConsumerElected() {
log.Printf("pos: replica %q is not the elected consumer, MQTT ingest idle here",
os.Getenv("HOSTNAME"))
return nil, nil
}
// Each ingest worker holds a database transaction while it runs, so the
// real ceiling is the Postgres connection pool rather than the CPU. The
// queue is deep enough to absorb a burst and shallow enough that a genuine
// overload is felt as backpressure rather than hidden as latency.
c := &PosMqttConsumer{
svc: svc,
ingest: newPosPool("ingest", posPoolSize("POS_INGEST_WORKERS", 8), 256),
health: newPosPool("health", posPoolSize("POS_HEALTH_WORKERS", 2), 512),
}
opts := mqtt.NewClientOptions().
AddBroker(url).
// Stable, so the broker resumes this session and redelivers anything
// in flight rather than treating every restart as a new subscriber.
// Defaults to the pod name so replicas can never collide: a second
// connection with the same client id evicts the first, and the two then
// fight in a reconnect loop that looks like a flapping network.
SetClientID(getEnvDefault("MQTT_CLIENT_ID",
getEnvDefault("HOSTNAME", "nearle-pos-ingest"))).
SetCleanSession(false).
// Ordered delivery keeps paho on one goroutine, which is what lets a
// full queue push back on the broker. With concurrent delivery paho
// would keep reading no matter how far behind the workers were.
SetOrderMatters(posOrderedDelivery).
SetAutoReconnect(true).
SetMaxReconnectInterval(30 * time.Second).
SetKeepAlive(30 * time.Second).
SetConnectionLostHandler(func(_ mqtt.Client, err error) {
log.Printf("pos: MQTT connection lost: %v", err)
})
if user := os.Getenv("MQTT_USER"); user != "" {
opts.SetUsername(user).SetPassword(os.Getenv("MQTT_PASSWORD"))
}
// Re-subscribed on every (re)connect rather than once at startup: with a
// broker that did not persist the session, a reconnect would otherwise come
// back silently subscribed to nothing.
opts.SetOnConnectHandler(func(client mqtt.Client) {
log.Printf("pos: connected to MQTT broker %s", url)
for topic, handler := range map[string]mqtt.MessageHandler{
topicOrders: wrapHandler(c.ingest, c.handleOrders),
topicCustomers: wrapHandler(c.ingest, c.handleCustomers),
topicHealth: wrapHandler(c.health, c.handleHealth),
} {
if token := client.Subscribe(topic, 1, handler); token.Wait() && token.Error() != nil {
log.Printf("pos: could not subscribe to %s: %v", topic, token.Error())
continue
}
log.Printf("pos: subscribed to %s", topic)
}
})
client := mqtt.NewClient(opts)
if token := client.Connect(); token.Wait() && token.Error() != nil {
return nil, fmt.Errorf("could not connect to the MQTT broker at %s: %w", url, token.Error())
}
c.client = client
return c, nil
}
func (c *PosMqttConsumer) handleOrders(_ mqtt.Client, msg mqtt.Message) {
var batch models.PosOrderBatch
if err := json.Unmarshal(msg.Payload(), &batch); err != nil {
// Dropped rather than retried: there is no batch id to answer with, and
// the till will time out and re-send anyway.
log.Printf("pos: discarding unreadable order batch on %s: %v", msg.Topic(), err)
return
}
store, terminal := topicIdentity(msg.Topic())
if batch.Storeid == "" {
batch.Storeid = store
}
if batch.Terminalid == "" {
batch.Terminalid = terminal
}
ack, err := c.svc.IngestOrders(batch)
if err != nil {
// Nothing committed, so nothing is acknowledged. The terminal keeps
// every bill and retries — which is the entire point of the design.
log.Printf("pos: order batch %s from %s/%s failed, not acking: %v",
batch.Batchid, store, terminal, err)
return
}
c.publishAck(store, terminal, ack)
log.Printf("pos: order batch %s from %s/%s — %d accepted, %d rejected",
batch.Batchid, store, terminal, len(ack.Accepted), len(ack.Rejected))
}
func (c *PosMqttConsumer) handleCustomers(_ mqtt.Client, msg mqtt.Message) {
var batch models.PosCustomerBatch
if err := json.Unmarshal(msg.Payload(), &batch); err != nil {
log.Printf("pos: discarding unreadable customer batch on %s: %v", msg.Topic(), err)
return
}
store, terminal := topicIdentity(msg.Topic())
if batch.Storeid == "" {
batch.Storeid = store
}
if batch.Terminalid == "" {
batch.Terminalid = terminal
}
ack, err := c.svc.IngestCustomers(batch)
if err != nil {
log.Printf("pos: customer batch %s from %s/%s failed, not acking: %v",
batch.Batchid, store, terminal, err)
return
}
c.publishAck(store, terminal, ack)
}
// handleHealth records one heartbeat.
//
// Never acknowledged. Presence is fire-and-forget: a till whose heartbeat
// failed must carry on selling, and a blank square on a dashboard is a far
// better outcome than a terminal that stopped because Redis was busy.
func (c *PosMqttConsumer) handleHealth(_ mqtt.Client, msg mqtt.Message) {
var health models.PosHealth
if err := json.Unmarshal(msg.Payload(), &health); err != nil {
log.Printf("pos: discarding unreadable heartbeat on %s: %v", msg.Topic(), err)
return
}
// From the topic, not the body — the same rule bills follow.
store, terminal := topicIdentity(msg.Topic())
if health.Locationid == "" {
health.Locationid = store
}
if health.Terminalid == "" {
health.Terminalid = terminal
}
// The broker's Last Will arrives here too, as a bare {"status":"offline"}
// with no other fields, which is exactly what should be recorded when a
// till loses power mid-shift.
if health.Status == "" {
health.Status = "online"
}
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
defer cancel()
if err := c.svc.RecordHealth(ctx, health); err != nil {
log.Printf("pos: could not record heartbeat from %s/%s: %v", store, terminal, err)
}
}
// publishAck answers the till that sent the batch, and only that till.
func (c *PosMqttConsumer) publishAck(store, terminal string, ack *models.PosAck) {
if store == "" || terminal == "" {
log.Printf("pos: cannot ack batch %s — the topic named no terminal", ack.Batchid)
return
}
payload, err := json.Marshal(ack)
if err != nil {
log.Printf("pos: could not encode ack for batch %s: %v", ack.Batchid, err)
return
}
topic := fmt.Sprintf("nearle/pos/%s/%s/ack", store, terminal)
// QoS 1: losing an ack means the till re-sends bills that are already
// banked. Harmless, because the ingest deduplicates — but wasted traffic on
// a shop line that may not have much to spare.
token := c.client.Publish(topic, 1, false, payload)
if !token.WaitTimeout(10*time.Second) || token.Error() != nil {
log.Printf("pos: could not publish ack to %s: %v", topic, token.Error())
}
}
// topicIdentity reads the store and terminal out of
// `nearle/pos/<store>/<terminal>/<kind>`.
//
// Taken from the topic rather than the body on purpose: a till that could name
// a store in its payload could post sales into another shop's books.
func topicIdentity(topic string) (store, terminal string) {
parts := strings.Split(topic, "/")
if len(parts) < 5 {
return "", ""
}
return parts[2], parts[3]
}
// PublishCatalogueChanged tells every till in a store to pull now.
//
// Retained, so a terminal that was switched off during the change still hears
// about it when it comes back.
func (c *PosMqttConsumer) PublishCatalogueChanged(storeID, revision string) error {
payload, err := json.Marshal(map[string]string{"revision": revision})
if err != nil {
return err
}
token := c.client.Publish(fmt.Sprintf("nearle/pos/%s/catalogue", storeID), 1, true, payload)
token.Wait()
return token.Error()
}
// Close disconnects, allowing a moment for in-flight acks to leave.
func (c *PosMqttConsumer) Close() {
if c == nil || c.client == nil {
return
}
// Workers drain before the connection closes, so a bill mid-commit still
// gets its ack out. Disconnecting first would strand it: committed here,
// unacknowledged there, and sent again on the till's next attempt.
c.ingest.stop()
c.health.stop()
quiesce, err := strconv.Atoi(getEnvDefault("MQTT_QUIESCE_MS", "2000"))
if err != nil || quiesce < 0 {
quiesce = 2000
}
c.client.Disconnect(uint(quiesce))
}
// posConsumerElected decides whether this replica runs the MQTT ingest.
//
// Rules, in order:
//
// - POS_MQTT_CONSUMER=always or =never settles it outright, for deployments
// that are not a StatefulSet or that want the consumer somewhere specific.
// - A StatefulSet pod name ending in `-0` is elected. Ordinals are stable and
// unique, so this needs no lock, no lease and no coordination.
// - Anything else — a bare container, a Deployment, local development —
// is elected, because a single instance that refused to consume would be a
// far more confusing failure than one that did.
func posConsumerElected() bool {
switch strings.ToLower(strings.TrimSpace(os.Getenv("POS_MQTT_CONSUMER"))) {
case "always", "true", "yes":
return true
case "never", "false", "no":
return false
}
host := strings.TrimSpace(os.Getenv("HOSTNAME"))
if i := strings.LastIndex(host, "-"); i >= 0 {
if ordinal := host[i+1:]; ordinal != "" && strings.Trim(ordinal, "0123456789") == "" {
// A StatefulSet ordinal. Only the first replica consumes.
return ordinal == "0"
}
}
// Not an ordinal-named pod, so there is nothing to elect against.
return true
}
func getEnvDefault(key, fallback string) string {
if v := strings.TrimSpace(os.Getenv(key)); v != "" {
return v
}
return fallback
}