import React, { useCallback, useEffect, useRef, useState } from 'react'; import { AlertTriangle, Ban, CheckCircle2, ClipboardCheck, Clock, FileSpreadsheet, Loader2, PlayCircle, RefreshCw, XCircle, ArrowRight, } from 'lucide-react'; import { api } from '../api/client'; import { ago, fmtBytes } from './pipelineFormat'; import { FileStages } from './pipelineShared'; /* * Admin -> Upload Results. * * What happened to the last file a colleague sent. No picking, no clicking: * the newest colleague upload is found and followed automatically, all eleven * stages, down to which rows the validation gate refused and why. * * THREE THINGS, IN THIS ORDER: the file's NAME, its ELEVEN STAGES, and the ROWS * THAT FAILED. That ordering is the brief, and it is why the run counters, the * "brands touched" line, the img/llm flag chips and the batch id were taken out * - each was true, none was the question being asked, and together they pushed * the filename down to eleven-point text in the middle of the card. What is * left besides those three is the handful of banners (queued, interrupted, * partial, storage error) that exist to explain a timeline which is entirely * grey; without them an interrupted run reads as a broken screen rather than as * a run waiting for a Resume. * * There is no history here and there is not meant to be one: exactly one run is * ever on screen, the newest, replaced in place as newer ones arrive. * * THIS TAB USED TO BE THE REVIEW INBOX, and the inbox is still here - see the * gate section below. It is just no longer the point of the screen, because * with UPLOAD_AUTORUN at its default of true a colleague's upload starts on * arrival and never waits for review, so the inbox is permanently empty and the * tab showed nothing but an empty state pointing at another tab. * * HOW THIS DIFFERS FROM DAGSTER ORCHESTRATION, which also draws stage * timelines: that screen is for a run YOU launched - you tick files and watch * the batch you started, including your own admin uploads. This one answers the * opposite question, about a run nobody here started, and it is the only place * that shows the per-row rejection reasons a sender needs to fix their sheet. * * WHICH RUN IS "THE COLLEAGUE'S". `submitted_by` is the discriminator and it is * exact, not a heuristic: POST /api/uploads/catalog always sets it (falling back * to the `sender` form label and then to "anonymous"), and the admin's own * POST /api/admin/catalog-batch/ingest never passes it at all, so it is null * there. A batch released from the inbox carries the original senders, which is * correct - it is still their upload, just started by hand. * * IT IS A PREFERENCE, NOT A GATE. It used to be a gate, and that made this tab * a dead end: a colleague who walked over and used the Batch Catalog Ingestion * tab instead of the API produced a run with a null `submitted_by`, which was * filtered out permanently however recent it was, and the screen said "No * colleague upload yet" - true of the field and false of the world. So when the * window holds no colleague upload we now adopt the newest run of any origin * and SAY SO, rather than showing nothing. The exactness above is what makes * that label honest: a null sender means the Admin UI started it. * * AND A FAILED LOOKUP IS NOT AN EMPTY ONE. Every discovery error used to be * swallowed into that same empty state, so an expired session, a 403, a dead * backend and a genuinely quiet week were one indistinguishable sentence. * `discoverError` now separates them. An error still never wipes a run already * on screen - a blip must not blank a report someone is reading - it annotates * it instead. */ // Two questions at two cadences. "Has anything new arrived?" is cheap and slow; // "how far has the run I am watching got?" is the one that needs to feel live, // and it stops itself at a terminal status so a tab left open overnight goes // quiet. Both are cleared on unmount. const DISCOVER_POLL_MS = 10000; const RUN_POLL_MS = 3000; // How far back to look for a colleague upload. The endpoint returns ALL runs // newest-first, admin batches included, so a window of 20 could be filled by // the admin's own uploads and hide the colleague's. Slim rows are not tiny - // they keep stages[], rejections and errors - so ask for 50 on the steady-state // tick and widen to the endpoint's ceiling only when nothing was found. const DISCOVER_LIMIT = 50; const DISCOVER_MAX_LIMIT = 100; // When to stop following a run. Wider than the Orchestration tab's set // (OrchestrationPanel.jsx), which omits both `retired` and `interrupted`. // // `interrupted` is not terminal on the backend - a Resume can still move it - // but nothing will move it on its own, so a 3s poll would hammer a batch that // cannot change until an operator acts. Park on it; the 10s discovery tick // notices if BATCH_AUTO_RESUME re-queues it after a restart. const SETTLED = new Set([ 'done', 'failed', 'partial', 'cancelled', 'retired', 'interrupted', ]); const STATUS_ICON = { done: CheckCircle2, partial: AlertTriangle, failed: XCircle, running: Loader2, interrupted: AlertTriangle, }; const STATUS_TONE = { done: 'text-leaf-600', partial: 'text-amber-600', failed: 'text-maroon-600', running: 'text-amber-600', interrupted: 'text-amber-600', }; /* Is this row a colleague's upload, and does it have anything to show? * * `submitted_by` alone is not enough. GET /batches drops only PENDING rows * (batch_catalog.py), so a drop that was dismissed outright survives the filter * as a RETIRED husk - it still carries the sender's name, but every file was * discarded, so it has no result and no stage timeline. Without the second * clause, dismissing a drop makes that husk "the latest upload" and the panel * renders eleven grey rows that will never fill in. * * A drop that was RELEASED rather than dismissed is harmless either way: the * run it produced is staged before the drop is retired, so the run is newer and * wins the ordering regardless. */ function isColleagueUpload(batch) { return ( typeof batch?.submitted_by === 'string' && batch.submitted_by.trim() !== '' && batch.status !== 'retired' ); } /* The fallback: any run at all, as long as it is not a retired husk. * * Same second clause as above and for the same reason - a dismissed drop keeps * its row but has no stages and no result, so adopting one would draw eleven * grey rows that can never fill in. `pending` needs no clause here: GET * /batches drops those server-side (batch_catalog.py). */ function isShowableRun(batch) { return Boolean(batch) && batch.status !== 'retired'; } function IssueList({ title, items, tone = 'slate' }) { if (!items.length) return null; const border = tone === 'maroon' ? 'border-maroon-500/25 bg-maroon-100/30' : 'border-ink-900/10 bg-white'; return (

{title} ({items.length})

); } export function UploadResultsPanel({ onBatchStarted }) { // --- the results half --------------------------------------------------- const [run, setRun] = useState(null); const [discovered, setDiscovered] = useState(false); // first discovery done // Why the panel is empty, when it is. '' means the last lookup succeeded. const [discoverError, setDiscoverError] = useState(''); // How many runs that lookup examined, so an empty screen can say "looked at // 40 runs, none of them showable" rather than an unqualified "nothing". const [scanned, setScanned] = useState(0); const runIdRef = useRef(null); // Whether the widened lookup has already been spent - see discover(). const escalatedRef = useRef(false); // --- the gate half (only rendered when something is actually waiting) ---- const [inbox, setInbox] = useState(null); const [selected, setSelected] = useState(() => new Set()); const [busy, setBusy] = useState(''); const [error, setError] = useState(''); const [notice, setNotice] = useState(''); const [useLlm, setUseLlm] = useState(true); const [fetchImages, setFetchImages] = useState(true); /* Find the run to show, and read the inbox in the same tick. * * A colleague upload is preferred; failing that, the newest run of any * origin, so the tab is never a dead end. See the header comment. * * A failed poll never replaces a run already on screen - a transient blip * must not wipe a report the admin is reading, and the next tick retries. * But it IS recorded in `discoverError`, and the two renderings differ: with * a run up it is a staleness strip, with nothing up it is the whole card. * Silently treating a failure as "no uploads" is the bug this tab had. * * The inbox read is kept separate and stays silent. It is the secondary half * of the screen and its failure must not blank the results. */ const discover = useCallback(async () => { try { let { batches } = await api.listCatalogBatches(DISCOVER_LIMIT); let latest = (batches || []).find(isColleagueUpload); // Nothing in the window. Before concluding there is no colleague upload, // widen once to the endpoint's ceiling - a burst of admin batches can push // a real one past 50. Guarded so it happens at most once per mount (and // once per manual Refresh), never on the steady-state path where a run is // already on screen. if (!latest && !escalatedRef.current) { escalatedRef.current = true; ({ batches } = await api.listCatalogBatches(DISCOVER_MAX_LIMIT)); latest = (batches || []).find(isColleagueUpload); } // No colleague upload in reach. Show the newest run there is rather than // an empty screen - see the header comment. The label that goes with it is // derived at render from `submitted_by`, which is exact. const chosen = latest || (batches || []).find(isShowableRun) || null; setScanned((batches || []).length); setDiscoverError(''); if (chosen && chosen.batch_id !== runIdRef.current) { runIdRef.current = chosen.batch_id; // Show the slim row immediately - it already carries stages[] and the // counters, because slim strips only result.products - then let the // detail poll replace it with the full record. setRun(chosen); } else if (!chosen) { runIdRef.current = null; setRun(null); } } catch (err) { // Deliberately NOT clearing `run`: a transient blip must not wipe a // report someone is reading. The strip below says it may be stale. setDiscoverError(err?.message || 'Could not read the list of ingestion runs.'); } try { setInbox(await api.listInbox()); } catch { /* next tick retries */ } setDiscovered(true); }, []); useEffect(() => { discover(); const timer = setInterval(discover, DISCOVER_POLL_MS); return () => clearInterval(timer); }, [discover]); // An explicit retry buys back the widened lookup, so an admin who knows a // colleague just uploaded can force the deep search. Shared by the header's // Refresh button and the error card below, which must behave identically. const retry = useCallback(() => { escalatedRef.current = false; discover(); }, [discover]); // Follow the watched run until it settles. Re-created whenever the id or the // status changes, so it stops itself the moment the run settles rather than // polling a finished batch forever. const runId = run?.batch_id; const runStatus = run?.status; useEffect(() => { if (!runId || SETTLED.has(runStatus)) return undefined; let alive = true; const tick = async () => { try { const full = await api.getCatalogBatch(runId); if (alive && full?.batch_id === runIdRef.current) setRun(full); } catch { /* next tick retries */ } }; tick(); const timer = setInterval(tick, RUN_POLL_MS); return () => { alive = false; clearInterval(timer); }; }, [runId, runStatus]); // A run that was already settled when we adopted it never enters the poll // above, so fetch its full record once. A run that settles while being // followed already had its last fetch from that poll. useEffect(() => { if (!runId || !SETTLED.has(runStatus)) return undefined; let alive = true; (async () => { try { const full = await api.getCatalogBatch(runId); if (alive && full?.batch_id === runIdRef.current) setRun(full); } catch { /* the slim row is already on screen; it is enough */ } })(); return () => { alive = false; }; // Deliberately keyed on the id alone: re-running this on every status // change would re-fetch a settled batch for no reason. // eslint-disable-next-line react-hooks/exhaustive-deps }, [runId]); // --- gate actions (unchanged behaviour, just conditionally rendered) ----- const toggle = (fileId) => { setSelected((current) => { const next = new Set(current); if (next.has(fileId)) next.delete(fileId); else next.add(fileId); return next; }); }; const toggleDrop = (submission) => { const ids = submission.files.map((f) => f.file_id); const allOn = ids.every((id) => selected.has(id)); setSelected((current) => { const next = new Set(current); ids.forEach((id) => (allOn ? next.delete(id) : next.add(id))); return next; }); }; const handleStart = async () => { if (!selected.size) return; setBusy('start'); setError(''); setNotice(''); try { const batch = await api.startBatchFromInbox([...selected], { use_llm: useLlm, fetch_images: fetchImages, }); setSelected(new Set()); setNotice( `Started ${batch.files_total} file(s) as one batch. Follow it on the ` + `Batch Catalog Ingestion tab.` ); await discover(); if (onBatchStarted) onBatchStarted(batch); } catch (err) { setError(err?.message || 'Could not start those files.'); await discover(); } finally { setBusy(''); } }; const handleDismiss = async () => { if (!selected.size) return; setBusy('dismiss'); setError(''); setNotice(''); try { const result = await api.dismissInboxFiles([...selected]); setSelected(new Set()); setNotice(`Dismissed ${result.dismissed} file(s). They will not be processed.`); await discover(); } catch (err) { setError(err?.message || 'Could not dismiss those files.'); await discover(); } finally { setBusy(''); } }; // Drop any selection that is no longer pending - another tab may have started // or dismissed it, and a checkbox pointing at a file that is already gone // would 409 on the next click with no explanation. useEffect(() => { if (!inbox) return; const live = new Set(inbox.submissions.flatMap((s) => s.files.map((f) => f.file_id))); setSelected((current) => new Set([...current].filter((id) => live.has(id)))); }, [inbox]); const pending = inbox?.pending_count ?? 0; const submissions = inbox?.submissions || []; // Exact, not a guess: only POST /api/admin/catalog-batch/ingest leaves the // sender null - uploads.py always sets one, and a release from the inbox // carries the original senders through. const fromColleague = isColleagueUpload(run); const files = run?.files || []; /* The answer to "what did they send?", promoted to the headline. * * A drop can carry several files and each still gets its own named row in the * timeline below, so naming the first and counting the rest loses nothing. * `current_file` is the fallback for the window where a batch exists but its * file list has not been rendered yet. */ const headlineName = files[0]?.filename || run?.current_file || 'file name unavailable'; const extraFiles = Math.max(0, files.length - 1); const rejections = files.flatMap((f) => (f.result?.rejections || []).map((r) => { const where = r.row == null ? 'row unknown' : `row ${r.row}`; const size = r.size ? ` ${r.size}` : ''; return `${where} · ${r.product_name || '(unnamed)'}${size} — ${r.reason}`; }) ); const rowErrors = files.flatMap((f) => (f.result?.errors || []).map( (e) => `row ${e.row} · ${e.product_name || '(unnamed)'} — ${e.error}` ) ); const warnings = files.flatMap((f) => (f.result?.warnings || []).map((w) => w)); const storageErrors = files .filter((f) => f.result?.storage_error) .map((f) => `${f.filename} — ${f.result.storage_error}`); const StatusIcon = STATUS_ICON[run?.status] || Clock; return (
{/* ---- the review gate, only when something is actually waiting ------ With UPLOAD_AUTORUN on this never renders. Turn the setting off and it comes back on its own, which is the behaviour settings.py promises - "the review inbox comes back with no code change". */} {pending > 0 && (

Waiting for review

UPLOAD_AUTORUN is off, so these have been held rather than run. Tick what you want and press Start selected to send them through the 11-stage pipeline as one batch. Files from different drops can go in the same batch.

{pending} file(s) awaiting review
{/* These govern the manual start above and nothing else. An upload that ran itself never passed through here - the server reads UPLOAD_AUTORUN_FETCH_IMAGES / UPLOAD_AUTORUN_USE_LLM and ignores whatever a request asked for. What a given run actually used is the img/llm pair on its own header below. */}
Applies to Start selected
{notice && (
{notice}
)} {error && (
{error}
)}
{submissions.map((submission) => { const ids = submission.files.map((f) => f.file_id); const allOn = ids.length > 0 && ids.every((id) => selected.has(id)); return (

from {submission.submitted_by} {ago(submission.created_at)} · {submission.files.length} file(s)

    {submission.files.map((file) => (
  • ))}
); })}
)} {/* ---- what the page is for ---------------------------------------- */}

Upload Results

The most recent spreadsheet a colleague sent to{' '} /api/uploads/catalog, and everything the 11-stage pipeline did with it — including which rows were refused and why. It updates on its own; there is nothing to pick. When no colleague upload is in reach it falls back to the most recent run of any kind, and says so, rather than showing you an empty screen.

{/* ---- the run ------------------------------------------------------ */} {!discovered && (
)} {/* Nothing on screen AND the lookup failed. This is the case that used to render as "No colleague upload yet", sending people to look for an upload that had arrived perfectly well - the request for it 401'd. */} {discovered && !run && discoverError && (

Could not read the ingestion runs

{discoverError}

This is not the same as having no uploads — nothing has been checked. If that reads as expired or forbidden, sign in again; otherwise the backend is unreachable.

)} {/* Gate open but nothing has run: "no upload yet" would be flatly false - there is one, it is sitting above waiting for a decision. */} {discovered && !run && !discoverError && pending > 0 && (

Nothing has run yet — the upload above is waiting for your approval. Tick the files and press Start selected, and its stage breakdown appears here.

)} {/* Genuinely nothing to show - and it now says WHICH nothing. With the any-origin fallback in discover(), `scanned > 0` survives only for the case where every run in reach is a retired husk. */} {discovered && !run && !discoverError && pending === 0 && (
{scanned === 0 ? (

No ingestion run at all yet. The moment a colleague sends a sheet to{' '} POST /api/uploads/catalog, its run appears here with the full stage breakdown.

) : (

Looked at the last {scanned} ingestion run(s) and none can be shown — every one of them was dismissed from the review inbox.

)}

Only the most recent {DISCOVER_MAX_LIMIT} runs are searched, and uploads older than a week are cleared by retention.

)} {run && (
{/* What is below is whatever the last SUCCESSFUL lookup found. Saying so is the point - showing it silently as live is how a dead backend passed for a quiet one. */} {discoverError && (
The last refresh failed ({discoverError}), so this may be out of date. Retrying every {Math.round(DISCOVER_POLL_MS / 1000)}s.
)} {/* No colleague upload was in reach, so this is the newest run of any origin. Never silent about it: the tab promises a colleague's upload, and a run with no sender is not one. */} {!fromColleague && (
No colleague upload in the last {scanned} ingestion run(s), so this is simply the most recent run. It carries no sender, which means it was started from the Admin UI (Batch Catalog Ingestion) rather than sent to{' '} POST /api/uploads/catalog. Its eleven stages are below all the same.
)} {/* The filename IS the headline - `truncate` on a min-w-0 flex child so a long name is cut rather than pushing the card sideways. */}

{headlineName}

{/* Everything that used to crowd the filename, demoted to one line. */}

{run.submitted_by ? ( from {run.submitted_by} ) : ( started from the Admin UI )} · {ago(run.created_at)} · {run.status} {extraFiles > 0 && ( <> · +{extraFiles} more file(s), listed below )}

{/* `partial` exists so this does not read as success: some files landed and some did not. */} {run.status === 'partial' && (
{run.files_done} of {run.files_total} file(s) landed; {run.files_failed} did not.
)} {run.status === 'queued' && (
{run.runner === 'dagster' ? 'Staged and waiting for the Dagster orchestrator to claim it. Dagster is not deployed in production, so this will wait indefinitely there.' : 'Staged and waiting its turn - the ingestion queue was busy when this arrived.'}
)} {run.status === 'interrupted' && (
The backend restarted while this was running, so its unfinished files were put back in the queue. Resume it from Batch Catalog Ingestion.
)} {run.detail && (

{run.detail}

)} {/* A storage error still leaves rows built, so it hides among the counters - but it is what forces the file to failed, so it is called out on its own. */} {storageErrors.length > 0 && (

Could not write to the catalogue

{storageErrors.map((text, i) => (

{text}

))}
)} {/* The third of the three things: a refused row, named, with the sheet line number the sender sees on screen (header counted as row 1). This is the only place in the app that reports them. */}
)}
); }