Workers hardcoded jupiter's ClusterIP (10.43.224.63 / 10.43.229.168), which had gone stale and pointed at nothing. Every request forwarded from worker-orders, worker-deliveries, worker-customers, worker-rider-logs, worker-products, and worker-notifications to jupiter was timing out silently, breaking order creation, delivery logs, and rider online status. Repointed at the stable in-cluster DNS name (jupiter.nearle) instead of a ClusterIP so this can't go stale again after a future service recreation. Also bumps jupiter to v2.7.57 (Redis client timeout/pool fix) to match what's already deployed live. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
123 lines
3.7 KiB
YAML
123 lines
3.7 KiB
YAML
# Reconstructed manifest - this StatefulSet was originally deployed
|
|
# manually outside of git, and was lost when the core namespace got
|
|
# recreated during the Flux removal incident on 2026-07-20. Rebuilt from
|
|
# its confirmed runtime config (NATS_STREAM/NATS_CONSUMER/FILTER_SUBJECT
|
|
# seen in its own logs) plus the same pattern as its sibling workers.
|
|
# Resource requests/limits and WORKER_CONCURRENCY are best-guess matches
|
|
# to the lightest sibling worker (worker-rider-logs) - adjust if the
|
|
# original values are known.
|
|
apiVersion: apps/v1
|
|
kind: StatefulSet
|
|
metadata:
|
|
name: worker-notifications
|
|
namespace: core
|
|
labels:
|
|
app.kubernetes.io/name: worker-notifications
|
|
app.kubernetes.io/component: worker
|
|
spec:
|
|
serviceName: "worker-notifications"
|
|
replicas: 1
|
|
selector:
|
|
matchLabels:
|
|
app.kubernetes.io/name: worker-notifications
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app.kubernetes.io/name: worker-notifications
|
|
app.kubernetes.io/component: worker
|
|
annotations:
|
|
prometheus.io/scrape: "true"
|
|
prometheus.io/port: "9090"
|
|
prometheus.io/path: "/metrics"
|
|
spec:
|
|
terminationGracePeriodSeconds: 45
|
|
tolerations:
|
|
- key: dedicated
|
|
operator: Equal
|
|
value: workers
|
|
effect: NoSchedule
|
|
affinity:
|
|
nodeAffinity:
|
|
requiredDuringSchedulingIgnoredDuringExecution:
|
|
nodeSelectorTerms:
|
|
- matchExpressions:
|
|
- key: node-role.workolik/worker
|
|
operator: In
|
|
values:
|
|
- "true"
|
|
podAntiAffinity:
|
|
preferredDuringSchedulingIgnoredDuringExecution:
|
|
- weight: 100
|
|
podAffinityTerm:
|
|
labelSelector:
|
|
matchExpressions:
|
|
- key: app.kubernetes.io/name
|
|
operator: In
|
|
values:
|
|
- worker-notifications
|
|
topologyKey: kubernetes.io/hostname
|
|
topologySpreadConstraints:
|
|
- maxSkew: 1
|
|
topologyKey: kubernetes.io/hostname
|
|
whenUnsatisfiable: ScheduleAnyway
|
|
labelSelector:
|
|
matchLabels:
|
|
app.kubernetes.io/name: worker-notifications
|
|
containers:
|
|
- name: worker
|
|
image: workolik360/nats-worker:v1.1.0
|
|
imagePullPolicy: IfNotPresent
|
|
securityContext:
|
|
allowPrivilegeEscalation: false
|
|
capabilities:
|
|
drop:
|
|
- ALL
|
|
command: ["python3", "-u", "/scripts/worker.py"]
|
|
volumeMounts:
|
|
- name: worker-script-vol
|
|
mountPath: /scripts
|
|
envFrom:
|
|
- configMapRef:
|
|
name: core-config
|
|
env:
|
|
- name: NATS_STREAM
|
|
value: "NOTIFICATIONS"
|
|
- name: NATS_CONSUMER
|
|
value: "notifications-worker"
|
|
- name: FILTER_SUBJECT
|
|
value: "api.v1.notifications.push"
|
|
- name: WORKER_CONCURRENCY
|
|
value: "10"
|
|
- name: NATS_USER
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: nats-credentials
|
|
key: username
|
|
- name: NATS_PASSWORD
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: nats-credentials
|
|
key: password
|
|
- name: EXTERNAL_ENDPOINT_API_KEY
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: external-endpoint-secrets
|
|
key: api_key
|
|
optional: true
|
|
- name: EXTERNAL_BASE_URL
|
|
value: "http://jupiter.nearle"
|
|
resources:
|
|
requests:
|
|
memory: "128Mi"
|
|
cpu: "40m"
|
|
limits:
|
|
memory: "128Mi"
|
|
cpu: "200m"
|
|
ports:
|
|
- containerPort: 9090
|
|
name: metrics
|
|
volumes:
|
|
- name: worker-script-vol
|
|
configMap:
|
|
name: worker-script
|