371 lines
13 KiB
TypeScript
371 lines
13 KiB
TypeScript
/**
|
|
* The review inbox, and the status that nearly slipped through as success.
|
|
*
|
|
* The fixture is the live response to an anonymous upload on 28 Aug 2026 —
|
|
* `status: "pending"`, every total zero, the file still queued.
|
|
*/
|
|
import assert from 'node:assert/strict';
|
|
import { test } from 'node:test';
|
|
import {
|
|
isAwaitingReview,
|
|
isDismissed,
|
|
isSettled,
|
|
pollDelayFor,
|
|
currentStage,
|
|
isStuckOnMissingRunner,
|
|
productsOf,
|
|
releasedRunId,
|
|
summarise,
|
|
type IngestBatch,
|
|
} from './ingest';
|
|
|
|
const held = {
|
|
batch_id: '2ee38d06b583454ea0278a7f6de2c87f',
|
|
status: 'pending',
|
|
detail: 'Waiting for review. Nothing runs until an admin starts it.',
|
|
submitted_by: 'anonymous',
|
|
files_total: 1,
|
|
files_done: 0,
|
|
files_failed: 0,
|
|
totals: {
|
|
rows_total: 0,
|
|
products_built: 0,
|
|
inserted: 0,
|
|
backfilled: 0,
|
|
skipped_existing: 0,
|
|
rejected: 0,
|
|
},
|
|
brands: [],
|
|
files: [{ index: 0, filename: 'qa.csv', status: 'queued' as const }],
|
|
} satisfies IngestBatch;
|
|
|
|
// The bug this guards: `isSettled` used to mean "not queued and not running",
|
|
// so `pending` counted as finished and the panel rendered a completed import of
|
|
// zero products for a batch that had not started.
|
|
test('a batch held for review is not treated as finished', () => {
|
|
assert.equal(isSettled(held), false, 'a held batch must not read as settled');
|
|
assert.equal(isAwaitingReview(held), true);
|
|
});
|
|
|
|
test('the summary says it is waiting, not that nothing imported', () => {
|
|
const line = summarise(held);
|
|
assert.match(line, /review/i);
|
|
assert.doesNotMatch(line, /0 added/, 'must not report an import that never ran');
|
|
});
|
|
|
|
test('a real result is still settled', () => {
|
|
for (const status of ['done', 'partial', 'failed', 'interrupted', 'cancelled'] as const) {
|
|
assert.equal(isSettled({ ...held, status }), true, `${status} should be settled`);
|
|
}
|
|
});
|
|
|
|
test('queued and running are still in flight', () => {
|
|
for (const status of ['queued', 'running'] as const) {
|
|
assert.equal(isSettled({ ...held, status }), false, `${status} should not be settled`);
|
|
}
|
|
});
|
|
|
|
// isSettled is written as a positive list precisely so a status nobody
|
|
// anticipated stalls a spinner rather than fabricating a completed import.
|
|
test('an unknown future status does not read as finished', () => {
|
|
const unknown = { ...held, status: 'quarantined' as unknown as IngestBatch['status'] };
|
|
assert.equal(isSettled(unknown), false);
|
|
});
|
|
|
|
/* ── The drop lifecycle ───────────────────────────────────────────────────── */
|
|
|
|
const released = {
|
|
...held,
|
|
batch_id: '9f088d949aa9',
|
|
status: 'pending' as const,
|
|
files: [{ index: 0, filename: 'qa.csv', status: 'released' as const, released_to: '8dcef8a2ad94' }],
|
|
} satisfies IngestBatch;
|
|
|
|
const dismissed = {
|
|
...held,
|
|
files: [{ index: 0, filename: 'qa.csv', status: 'dismissed' as const, released_to: null }],
|
|
} satisfies IngestBatch;
|
|
|
|
// A released drop is not still waiting — the run is one hop away, and treating
|
|
// it as held would leave the screen saying "queued for review" forever.
|
|
test('a released drop is no longer awaiting review', () => {
|
|
assert.equal(releasedRunId(released), '8dcef8a2ad94');
|
|
assert.equal(isAwaitingReview(released), false);
|
|
assert.equal(isAwaitingReview(held), true, 'an unreleased drop is still waiting');
|
|
});
|
|
|
|
// Declined is terminal. Polling on is waiting for something that cannot happen.
|
|
test('a dismissed drop is recognised and reported as declined', () => {
|
|
assert.equal(isDismissed(dismissed), true);
|
|
assert.equal(isDismissed(held), false);
|
|
assert.match(summarise(dismissed), /declined/i);
|
|
assert.doesNotMatch(summarise(dismissed), /0 added/);
|
|
});
|
|
|
|
test('the manifest is collected across files', () => {
|
|
const done = {
|
|
...held,
|
|
status: 'done' as const,
|
|
files: [
|
|
{
|
|
index: 0,
|
|
filename: 'a.csv',
|
|
status: 'done' as const,
|
|
result: {
|
|
products: [
|
|
{
|
|
image_id: 'amul_amul_butter_100g',
|
|
brand: 'amul',
|
|
product_name: 'Amul Butter 100g',
|
|
product_sku: 'ACME-BUT-100',
|
|
sku_source: 'sheet',
|
|
disposition: 'inserted' as const,
|
|
},
|
|
],
|
|
},
|
|
},
|
|
],
|
|
} satisfies IngestBatch;
|
|
|
|
const products = productsOf(done);
|
|
assert.equal(products.length, 1);
|
|
// image_id is the join key; matching on name creates duplicates instead of
|
|
// updating, which is why it is asserted rather than the name.
|
|
assert.equal(products[0]?.image_id, 'amul_amul_butter_100g');
|
|
assert.equal(products[0]?.disposition, 'inserted');
|
|
});
|
|
|
|
/* ── retired: the drop is spent, the answer is on the files ───────────────── */
|
|
|
|
// A drop released into a run reads `retired`, and the run is elsewhere. Calling
|
|
// it finished would report an import that is running right now as a completed
|
|
// import of zero products.
|
|
test('a retired drop that was released is not finished', () => {
|
|
const retired = {
|
|
...held,
|
|
status: 'retired' as const,
|
|
files: [
|
|
{ index: 0, filename: 'qa.csv', status: 'released' as const, released_to: '8dcef8a2ad94' },
|
|
],
|
|
} satisfies IngestBatch;
|
|
|
|
assert.equal(isSettled(retired), false, 'the run still has to be followed');
|
|
assert.equal(releasedRunId(retired), '8dcef8a2ad94');
|
|
});
|
|
|
|
// Retired with nothing to follow is genuinely over — otherwise the panel spins
|
|
// on a drop that no longer exists.
|
|
test('a retired drop with nowhere to follow is finished', () => {
|
|
const retired = {
|
|
...held,
|
|
status: 'retired' as const,
|
|
files: [{ index: 0, filename: 'qa.csv', status: 'dismissed' as const, released_to: null }],
|
|
} satisfies IngestBatch;
|
|
|
|
assert.equal(isSettled(retired), true);
|
|
});
|
|
|
|
/* ── Cross-drop contamination ─────────────────────────────────────────────── */
|
|
|
|
// An admin can assemble one run from several drops, so a run's manifest can
|
|
// carry other senders' products. Applying our sheet's price and opening stock
|
|
// to those would stock someone else's goods into our merchant's branch.
|
|
test('only our own file contributes products', () => {
|
|
const run = {
|
|
...held,
|
|
status: 'done' as const,
|
|
files: [
|
|
{
|
|
index: 0,
|
|
filename: 'ours.csv',
|
|
status: 'done' as const,
|
|
result: {
|
|
products: [
|
|
{ image_id: 'amul_a', brand: 'amul', product_name: 'Ours', disposition: 'inserted' as const },
|
|
],
|
|
},
|
|
},
|
|
{
|
|
index: 1,
|
|
filename: 'someone-elses.csv',
|
|
status: 'done' as const,
|
|
result: {
|
|
products: [
|
|
{ image_id: 'amul_b', brand: 'amul', product_name: 'Theirs', disposition: 'inserted' as const },
|
|
],
|
|
},
|
|
},
|
|
],
|
|
} satisfies IngestBatch;
|
|
|
|
const mine = productsOf(run, ['ours.csv']);
|
|
assert.equal(mine.length, 1);
|
|
assert.equal(mine[0]?.product_name, 'Ours');
|
|
|
|
// Unfiltered still returns everything — the filter is the caller's decision,
|
|
// and every caller that prices products must make it.
|
|
assert.equal(productsOf(run).length, 2);
|
|
});
|
|
|
|
/*
|
|
The stage timeline, and the runner that silently isn't there.
|
|
|
|
Both arrived with the ingest team's 31 Aug documentation update. The timeline is
|
|
what lets the console draw the real eleven stages instead of a file-count bar;
|
|
the runner is a trap, and the more important of the two.
|
|
*/
|
|
|
|
const runningFile = {
|
|
index: 0,
|
|
filename: 'catalog.csv',
|
|
status: 'running' as const,
|
|
stage_index: 6,
|
|
stage_name: 'Image Search & Contamination Filtering',
|
|
total_stages: 11,
|
|
rows_done: 120,
|
|
rows_total: 400,
|
|
stages: [
|
|
{
|
|
index: 1,
|
|
name: 'Brand Resolution & FSSAI Licence Mapping',
|
|
rows_done: 400,
|
|
rows_total: 400,
|
|
started_at: 1756612800.1,
|
|
finished_at: 1756612801.4,
|
|
},
|
|
{
|
|
index: 6,
|
|
name: 'Image Search & Contamination Filtering',
|
|
rows_done: 120,
|
|
rows_total: 400,
|
|
started_at: 1756612809.7,
|
|
finished_at: null,
|
|
},
|
|
],
|
|
};
|
|
|
|
// `finished_at: null` is the marker, not the last array entry and not
|
|
// `stage_index`. Reading the position any other way breaks the moment a stage
|
|
// completes out of order or the array carries a trailing finished entry.
|
|
test('the running stage is the one with no finish time', () => {
|
|
const stage = currentStage(runningFile);
|
|
assert.equal(stage?.index, 6);
|
|
assert.equal(stage?.rows_done, 120);
|
|
});
|
|
|
|
// A finished file keeps its history, which is the whole reason the timeline
|
|
// exists — the scalars only ever describe the present moment, and for a
|
|
// finished file that moment is over.
|
|
test('a finished file still reports its last stage', () => {
|
|
const done = {
|
|
...runningFile,
|
|
status: 'done' as const,
|
|
stages: runningFile.stages.map((s) => ({ ...s, finished_at: s.finished_at ?? 1756612900.0 })),
|
|
};
|
|
assert.equal(currentStage(done)?.index, 6);
|
|
});
|
|
|
|
// A service build that predates the timeline still has to render. The scalars
|
|
// are the fallback, not the source of truth.
|
|
test('a response without a timeline falls back to the scalars', () => {
|
|
const { stages: _stages, ...noTimeline } = runningFile;
|
|
const stage = currentStage(noTimeline);
|
|
assert.equal(stage?.index, 6);
|
|
assert.equal(stage?.name, 'Image Search & Contamination Filtering');
|
|
});
|
|
|
|
test('a file that has not started reports no stage at all', () => {
|
|
assert.equal(currentStage({ index: 0, filename: 'a.csv', status: 'queued' }), null);
|
|
});
|
|
|
|
/*
|
|
`runner: "dagster"` never runs in production — Dagster is a development tool,
|
|
absent from the deployed image — so the batch waits for a worker that will never
|
|
claim it. Every visible signal is identical to a batch merely waiting its turn,
|
|
which is exactly why it has to be named rather than rendered as progress.
|
|
*/
|
|
test('a batch staged for the absent orchestrator is called out', () => {
|
|
assert.equal(isStuckOnMissingRunner({ ...held, status: 'queued', runner: 'dagster' }), true);
|
|
});
|
|
|
|
test('the in-process runner is not a stall', () => {
|
|
assert.equal(isStuckOnMissingRunner({ ...held, status: 'queued', runner: 'inprocess' }), false);
|
|
});
|
|
|
|
// A batch that reached `running` plainly found an executor, whatever it was
|
|
// staged for. Warning then would contradict the progress on screen.
|
|
test('a batch already running is not stuck, whatever it was staged for', () => {
|
|
assert.equal(isStuckOnMissingRunner({ ...held, status: 'running', runner: 'dagster' }), false);
|
|
});
|
|
|
|
/*
|
|
A drop is not a run, and the difference is easy to lose.
|
|
|
|
`released_to` lives on the DROP's files. Once an admin releases it, following
|
|
that pointer lands on the run — and the run carries no `released_to` of its own,
|
|
because nothing released it. So a caller who resolves first and asks for the run
|
|
id second gets null, and the only pointer from the id they hold to the id with
|
|
the results is never recorded.
|
|
|
|
This cost a real bug in both directions: the Uploads page never saved a run id,
|
|
and the import panel keyed the shelving write on `batch.batch_id` — which by
|
|
then was the run — updating a receipt row that does not exist, silently.
|
|
*/
|
|
test('the run id is on the drop, and gone from the run it points to', () => {
|
|
const drop = {
|
|
...held,
|
|
batch_id: 'drop-1',
|
|
status: 'retired' as const,
|
|
files: [
|
|
{ index: 0, filename: 'catalog.csv', status: 'released' as const, released_to: 'run-1' },
|
|
],
|
|
};
|
|
assert.equal(releasedRunId(drop), 'run-1');
|
|
|
|
// The same question asked of the run answers null. Read the drop first.
|
|
const run = {
|
|
...held,
|
|
batch_id: 'run-1',
|
|
status: 'done' as const,
|
|
files: [{ index: 0, filename: 'catalog.csv', status: 'done' as const }],
|
|
};
|
|
assert.equal(releasedRunId(run), null);
|
|
|
|
// And the two ids differ, which is exactly why a receipt keyed on the drop
|
|
// cannot be written using the run's.
|
|
assert.notEqual(drop.batch_id, run.batch_id);
|
|
});
|
|
|
|
/* ── How often to look, and when to stop looking ──────────────────────────── */
|
|
|
|
/*
|
|
The console used to stop polling the moment a drop went to review, on the
|
|
reasoning that waiting for an admin is not progress. It is not — but the release
|
|
IS, and stopping there meant the panel said "waiting for review" until somebody
|
|
reloaded the page. A step-by-step panel that only advances on reload is the
|
|
thing the panel exists to replace.
|
|
*/
|
|
|
|
test('a review hold is polled slowly, not abandoned', () => {
|
|
assert.equal(isAwaitingReview(held), true);
|
|
assert.equal(pollDelayFor(held), 15000, 'a hold can last hours; 2s would be 1,800 reads an hour');
|
|
});
|
|
|
|
test('a running batch is polled at a pace a person can watch', () => {
|
|
const running = { ...held, status: 'running' } satisfies IngestBatch;
|
|
assert.equal(isAwaitingReview(running), false);
|
|
assert.equal(pollDelayFor(running), 2000);
|
|
});
|
|
|
|
// A released drop is no longer waiting on anybody, so it goes back to the fast
|
|
// cadence even though its own status still reads "pending".
|
|
test('a released drop is followed at the running pace', () => {
|
|
const released = {
|
|
...held,
|
|
files: [{ index: 0, filename: 'qa.csv', status: 'queued' as const, released_to: 'run-77' }],
|
|
} satisfies IngestBatch;
|
|
assert.equal(releasedRunId(released), 'run-77');
|
|
assert.equal(isAwaitingReview(released), false);
|
|
assert.equal(pollDelayFor(released), 2000);
|
|
});
|