Files
doormilxpress_astryx/src/lib/assistant/scan.js

191 lines
7.7 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { getBookingsPage } from '@/api/doormile';
import { parseDoormileTimestamp } from '@/lib/doormileTimestamp';
/**
* Reading bookings for the assistant.
*
* `GET /admin/bookings` caps `pagesize` at 100 server-side — NOT 1000, which is
* what this comment claimed for a long time and what `express-console-api.md`
* documented. A page is 100 rows, so MAX_PAGES x BULK_PAGESIZE is not the
* reachable total: 12 pages is 1200 rows, not 12,000.
*
* It IS tenant-scoped, and it does accept `?status=` — but as an exact,
* case-sensitive, single-value match against the stored capitalised enum
* (`Pending_Pickup`). The status GROUPS this console renders as tabs each cover
* several values in lowercase, so the server filter cannot express a tab and
* the grouping still happens here. There is no date filter.
*
* The list is ordered `bookingid DESC`, so page 1 is the newest page and a
* drain that stops early keeps the most recent bookings rather than the oldest.
*
* Every scan returns `{ rows, truncated, scanned, pagesFetched, total }` rather
* than a bare array, and **`truncated` is not optional to handle**. A count
* built on a capped scan is a floor, not a total — run it through `countPhrase`,
* append `truncationNote` to the detail, and build the audit entry with
* `scanCall`, which reports an error rather than a green tick beside a partial
* number.
*/
const BULK_PAGESIZE = 1000;
/** 12k rows — generous, but bounded so one question cannot hammer the API. */
const MAX_PAGES = 12;
/**
* Row order is not documented. When page 1 comes back newest-first the scan can
* stop as soon as a page ends older than the window; otherwise it scans to the
* budget. Detected per call rather than assumed, so a backend change degrades to
* "scan everything" — slower, still correct — rather than to a wrong answer.
*/
const isDescendingByCreatedAt = (rows) => {
if (rows.length < 2) return false;
const first = parseDoormileTimestamp(rows[0].createdat);
const last = parseDoormileTimestamp(rows[rows.length - 1].createdat);
return first.isValid() && last.isValid() && first.valueOf() > last.valueOf();
};
const inRange = (booking, start, end) => {
const at = parseDoormileTimestamp(booking.createdat);
if (!at.isValid()) return false;
const day = at.format('YYYY-MM-DD');
return day >= start && day <= end;
};
/**
* A short-lived page cache.
*
* A multi-part question runs several intents, and a comparison runs two ranges —
* each would otherwise re-drain the same pages. Deliberately small and local:
* this is not a caching layer to settle on.
*/
const PAGE_CACHE_TTL_MS = 20_000;
const pageCache = new Map();
const getPageCached = (page) => {
const hit = pageCache.get(page);
if (hit && Date.now() - hit.at < PAGE_CACHE_TTL_MS) return hit.promise;
const promise = getBookingsPage(page, BULK_PAGESIZE).catch((err) => {
/* Never cache a failure — the next question should retry, not inherit it. */
pageCache.delete(page);
throw err;
});
pageCache.set(page, { at: Date.now(), promise });
return promise;
};
/**
* Drain pages up to the budget and return every row, unfiltered.
*
* `makeStop` receives page 1's rows once and returns the per-page early-stop
* predicate, so a range scan can bail as soon as it has read past its window
* while an unfiltered drain simply reads to the budget.
*
* `fetchPage` is injectable because the page cache below is the wrong layer for
* a caller that already has one. A TanStack-managed screen sets its own refetch
* interval, and serving it from a 20s module-level cache would silently cap how
* fresh that screen can ever be.
*/
const drainPages = async (makeStop, fetchPage = getPageCached) => {
const firstPage = await fetchPage(1);
const total = firstPage.total;
const pageSize = Math.max(1, firstPage.rows?.length || 100);
const pageCount = Math.max(1, Math.ceil(total / pageSize));
const budget = Math.min(pageCount, MAX_PAGES);
const collected = [...(firstPage.rows || [])];
const shouldStop = makeStop ? makeStop(firstPage.rows) : () => false;
let stoppedEarly = shouldStop(firstPage.rows) || collected.length >= total;
let lastPageFetched = 1;
for (let page = 2; page <= budget && !stoppedEarly; page += 1) {
const next = await fetchPage(page);
lastPageFetched = page;
if (!next.rows?.length) {
stoppedEarly = true;
break;
}
collected.push(...next.rows);
if (collected.length >= total) {
break;
}
stoppedEarly = shouldStop(next.rows);
}
return {
rows: collected,
/* Truncated only if the budget ran out with pages still unread AND the scan
did not stop early because it had already passed the window. */
truncated: !stoppedEarly && pageCount > budget,
scanned: collected.length,
pagesFetched: lastPageFetched,
total,
};
};
export const fetchBookingsInRange = async (start, end) => {
const scan = await drainPages((firstRows) => {
const descending = isDescendingByCreatedAt(firstRows);
/* Newest-first and this page already ends before the window opens → every
later page is older still, so there is nothing left to find. */
return (rows) => {
if (!descending || !rows.length) return false;
const oldest = parseDoormileTimestamp(rows[rows.length - 1].createdat);
return oldest.isValid() && oldest.format('YYYY-MM-DD') < start;
};
});
/* `scanned` deliberately stays the pre-filter count — it describes how much of
the account was read, which is what `truncated` has to be judged against. */
return { ...scan, rows: scan.rows.filter((booking) => inRange(booking, start, end)) };
};
/**
* Every booking the account has, up to the page budget, with **no date filter**.
*
* For screens that show the whole list rather than a window. Deliberately not
* `fetchBookingsInRange` with sentinel bounds: that path still runs `inRange`,
* which drops any row whose `createdat` will not parse. Dropping a row from a
* date-scoped answer is defensible; dropping it from "every order" is not — the
* row exists and the operator has to be able to see it.
*/
export const drainBookings = ({ cached = true } = {}) =>
drainPages(undefined, cached ? getPageCached : (page) => getBookingsPage(page, BULK_PAGESIZE));
export const fetchBookingsForDay = (day) => fetchBookingsInRange(day, day);
/**
* A full scan with no date window — for a per-order lookup, which has to look
* everywhere rather than inside a range. The sentinel bounds keep one code path:
* they can never trigger the early stop, so this always drains to the page
* budget and reports `truncated` honestly if the id could be further back.
*/
export const scanBookings = () => fetchBookingsInRange('0000-01-01', '9999-12-31');
/** A count over a truncated scan is a floor — say so. */
export const countPhrase = (scan, n) => `${scan.truncated ? 'At least ' : ''}${n}`;
export const truncationNote = (scan) =>
scan.truncated
? `\n\nScanned the most recent ${scan.scanned.toLocaleString('en-IN')} of ${scan.total.toLocaleString(
'en-IN'
)} bookings — this is a floor, not a complete count.`
: '';
/**
* The audit entry for a scan. Reports the real page count, and flags itself as
* an error when truncated so the sources strip cannot show a green "complete"
* beside a partial number.
*/
export const scanCall = (scan, note) => ({
name: 'getBookingsPage',
target: `/admin/bookings (${scan.pagesFetched} page${scan.pagesFetched === 1 ? '' : 's'} × ${BULK_PAGESIZE})`,
status: scan.truncated ? 'error' : 'complete',
errorMessage: scan.truncated ? `Scan capped at ${MAX_PAGES} pages; ${scan.total} bookings exist` : undefined,
stats: note,
});
/** A plain audit entry for a non-scan call. */
export const call = (name, target, stats) => ({ name, target, status: 'complete', stats });