updates on the ai agents and time series prediction and updates on the api to
This commit is contained in:
137
scratch/prediction_readiness.sql
Normal file
137
scratch/prediction_readiness.sql
Normal file
@@ -0,0 +1,137 @@
|
||||
-- Doormile — prediction readiness queries (Phase 0 + rung 1.0)
|
||||
-- docs/prediction-plan.md §2 and §3
|
||||
--
|
||||
-- ALL READ-ONLY. No DDL, no writes. Safe to run against production.
|
||||
-- Run top to bottom; each block prints one answer the plan needs.
|
||||
--
|
||||
-- Two of these depend on the consignment_booking view that migrate.go now
|
||||
-- creates (hazard H4 — consignments has no bookingid column). If the view does
|
||||
-- not exist yet, the blocks marked [needs view] will error; everything else
|
||||
-- still answers.
|
||||
|
||||
\echo '=== Q1. Is routing even running? (gates rung 1.1) ==='
|
||||
-- with_routed_eta = 0 means ROUTE_OPTIMIZER_URL is unset in the cluster and
|
||||
-- there is nothing to calibrate yet. See Phase 7 Track A1.
|
||||
SELECT count(*) AS assignments,
|
||||
count(*) FILTER (WHERE etaminutes > 0) AS with_routed_eta,
|
||||
count(*) FILTER (WHERE step > 0) AS sequenced,
|
||||
min(sequencedat) AS first_sequenced,
|
||||
max(sequencedat) AS last_sequenced
|
||||
FROM bookingassignments;
|
||||
|
||||
\echo '=== Q2. How much delivery history exists? (gates everything) ==='
|
||||
SELECT count(*) AS delivered_events,
|
||||
min(createdat)::date AS first_day,
|
||||
max(createdat)::date AS last_day,
|
||||
count(DISTINCT createdat::date) AS distinct_days
|
||||
FROM consignmenthistory
|
||||
WHERE eventstatus = 'Delivered';
|
||||
|
||||
\echo '=== Q3. H2 — delivered consignments with no Delivered event (silent label loss) ==='
|
||||
-- Non-zero means the training label is missing for those rows. Fix the write
|
||||
-- path before fitting anything.
|
||||
SELECT count(*) AS delivered_without_event
|
||||
FROM consignments c
|
||||
LEFT JOIN consignmenthistory h
|
||||
ON h.consignmentid = c.consignmentid AND h.eventstatus = 'Delivered'
|
||||
WHERE c.status = 'Delivered' AND h.consignmentid IS NULL;
|
||||
|
||||
\echo '=== Q4. H4 — can every consignment be resolved to a booking? [needs view] ==='
|
||||
SELECT count(*) AS consignments,
|
||||
count(*) FILTER (WHERE bookingid IS NULL) AS unresolvable
|
||||
FROM consignment_booking;
|
||||
|
||||
\echo '=== Q5. Per-pincode series length (gates demand grain) ==='
|
||||
-- Demand SARIMA/Prophet needs ~90 days PER SERIES, not in total.
|
||||
-- If days_with_data is small everywhere, aggregate to city level.
|
||||
SELECT c.deliverypincode,
|
||||
count(DISTINCT h.createdat::date) AS days_with_data,
|
||||
count(*) AS deliveries
|
||||
FROM consignmenthistory h
|
||||
JOIN consignments c ON c.consignmentid = h.consignmentid
|
||||
WHERE h.eventstatus = 'Delivered'
|
||||
GROUP BY 1
|
||||
ORDER BY 3 DESC
|
||||
LIMIT 30;
|
||||
|
||||
\echo '=== Q6. Rung 1.0 — how wrong are the promise constants? [needs view] ==='
|
||||
-- Actual duration uses consignmenthistory.createdat - consignments.createdat.
|
||||
-- Both come from CURRENT_TIMESTAMP defaults, so they are consistent with each
|
||||
-- other and this comparison is unaffected by H1.
|
||||
--
|
||||
-- Deliberately NOT comparing against estimateddeliveryat: that column is
|
||||
-- written with time.Now() at adminController.go:2738 but CURRENT_TIMESTAMP
|
||||
-- elsewhere, so its tagging is not uniform. Promise hours are taken from the
|
||||
-- service type instead, which is unambiguous.
|
||||
SELECT so.servicetype,
|
||||
count(*) AS n,
|
||||
round(avg(EXTRACT(EPOCH FROM (h.createdat - c.createdat))/3600)::numeric, 2) AS actual_hours_avg,
|
||||
round(percentile_cont(0.5) WITHIN GROUP (
|
||||
ORDER BY EXTRACT(EPOCH FROM (h.createdat - c.createdat))/3600)::numeric, 2) AS actual_hours_p50,
|
||||
round(percentile_cont(0.8) WITHIN GROUP (
|
||||
ORDER BY EXTRACT(EPOCH FROM (h.createdat - c.createdat))/3600)::numeric, 2) AS actual_hours_p80,
|
||||
round(percentile_cont(0.95) WITHIN GROUP (
|
||||
ORDER BY EXTRACT(EPOCH FROM (h.createdat - c.createdat))/3600)::numeric, 2) AS actual_hours_p95
|
||||
FROM consignments c
|
||||
JOIN consignmenthistory h ON h.consignmentid = c.consignmentid AND h.eventstatus = 'Delivered'
|
||||
JOIN consignment_booking cb ON cb.consignmentid = c.consignmentid
|
||||
LEFT JOIN bookingserviceoptions so ON so.bookingid = cb.bookingid
|
||||
GROUP BY 1
|
||||
ORDER BY 2 DESC;
|
||||
|
||||
\echo '=== Q7. Rung 1.0 — SLA breach rate against the promise actually stored ==='
|
||||
SELECT count(*) AS delivered,
|
||||
count(*) FILTER (WHERE c.sladueat IS NOT NULL) AS with_sla,
|
||||
count(*) FILTER (WHERE h.createdat > c.sladueat) AS breached
|
||||
FROM consignments c
|
||||
JOIN consignmenthistory h ON h.consignmentid = c.consignmentid AND h.eventstatus = 'Delivered';
|
||||
|
||||
\echo '=== Q8. H3 — do delivery durations look fabricated? ==='
|
||||
-- Real durations spread. Heavy clustering on round values, or a near-constant
|
||||
-- offset from createdat, means the timestamps are generated rather than
|
||||
-- observed. A model fitted on these predicts confidently and wrongly.
|
||||
SELECT round((EXTRACT(EPOCH FROM (h.createdat - c.createdat))/3600)::numeric, 0) AS duration_hours,
|
||||
count(*) AS n
|
||||
FROM consignments c
|
||||
JOIN consignmenthistory h ON h.consignmentid = c.consignmentid AND h.eventstatus = 'Delivered'
|
||||
GROUP BY 1
|
||||
ORDER BY 2 DESC
|
||||
LIMIT 20;
|
||||
|
||||
\echo '=== Q9. Demand series at city grain (the safe starting grain) ==='
|
||||
SELECT left(pickuppincode, 3) AS zone,
|
||||
count(DISTINCT createdat::date) AS days_with_data,
|
||||
count(*) AS bookings,
|
||||
min(createdat)::date AS first_day,
|
||||
max(createdat)::date AS last_day
|
||||
FROM pickupbookings
|
||||
WHERE status <> 'Cancelled'
|
||||
GROUP BY 1
|
||||
ORDER BY 3 DESC;
|
||||
|
||||
\echo '=== Q10. Would the calibration have any cells to fill? [needs view] ==='
|
||||
-- Mirrors the GROUP BY that internal/prediction/calibration.go uses. Cells
|
||||
-- below the sample floor are discarded, so this says whether rung 1.1 can
|
||||
-- calibrate at all or must stay on its fallback.
|
||||
-- Mirrors internal/prediction/calibration.go exactly, including the DISTINCT ON
|
||||
-- that picks ONE assignment per booking. A booking passes through several
|
||||
-- bookingassignments rows (Assigned, Rejected, Reassigned), so a plain join
|
||||
-- multiplies each delivery by its assignment history and overstates samples.
|
||||
WITH assigned AS (
|
||||
SELECT DISTINCT ON (ba.bookingid) ba.bookingid, ba.etaminutes
|
||||
FROM bookingassignments ba
|
||||
WHERE ba.etaminutes > 0
|
||||
ORDER BY ba.bookingid, ba.sequencedat DESC NULLS LAST, ba.bookingassignmentid DESC
|
||||
)
|
||||
SELECT left(c.deliverypincode, 3) AS zone,
|
||||
(EXTRACT(HOUR FROM c.createdat)::int / 3) AS hour_bucket,
|
||||
EXTRACT(ISODOW FROM c.createdat)::int AS weekday,
|
||||
count(*) AS samples
|
||||
FROM consignments c
|
||||
JOIN consignmenthistory h ON h.consignmentid = c.consignmentid AND h.eventstatus = 'Delivered'
|
||||
JOIN consignment_booking cb ON cb.consignmentid = c.consignmentid
|
||||
JOIN assigned a ON a.bookingid = cb.bookingid
|
||||
GROUP BY 1, 2, 3
|
||||
HAVING count(*) >= 20
|
||||
ORDER BY 4 DESC
|
||||
LIMIT 25;
|
||||
Reference in New Issue
Block a user