-- ============================================================================ -- One-off backfill: job_applications.worker_profile_id -- -- NOT A MIGRATION, and deliberately not one. `worker_profile_id` has existed -- since migration 000001 and the column is unchanged; what was missing was the -- frontend setting it when it filed an application from the talent pool. That -- is fixed going forward. This connects the rows written before the fix. -- -- A migration would make this part of the schema's history and run once per -- environment whether or not it was wanted. A script runs when somebody decides -- to run it, against the environment they name, after reading what it will do. -- -- WHAT IT MATCHES -- -- org_id AND case-insensitive email -- -- and nothing else. No fuzzy matching, no name comparison, no "probably the -- same person". Email is the identity `worker_profiles` itself already asserts: -- -- CREATE UNIQUE INDEX worker_profiles_org_email_key -- ON worker_profiles (org_id, email); -- -- That constraint is why an ambiguous match is not merely unlikely but -- IMPOSSIBLE: at most one worker profile can exist for any (org_id, email), so -- the join below can never find two. The HAVING clause asserts it anyway — -- cheap, and it fails loudly rather than silently picking one if that -- constraint is ever relaxed. -- -- `email` is `citext`, so the comparison is already case-insensitive; it is -- written with lower() so the intent survives a column type change. -- -- WHAT IT TOUCHES -- -- job_applications.worker_profile_id ONLY, and only where it is NULL. -- -- No status, no score, no timestamp, no other column, no other table. It never -- creates a worker_profiles row — an application with nobody behind it stays -- unlinked, which is the honest answer. -- -- SAFE TO RE-RUN. Rows already linked are excluded, so a second run changes -- nothing. -- -- USAGE -- psql -d -f scripts/backfill_worker_profile_id.sql -- -- Wrapped in a transaction: the report and the update see the same rows, and a -- failure leaves nothing half-applied. -- ============================================================================ BEGIN; \echo '' \echo '── Before ─────────────────────────────────────────────────────────────' WITH m AS ( SELECT a.worker_profile_id AS current_link, (SELECT count(*) FROM worker_profiles w WHERE w.org_id = a.org_id AND lower(w.email::text) = lower(a.email::text)) AS candidates FROM job_applications a ) SELECT count(*) AS applications, count(*) FILTER (WHERE current_link IS NOT NULL) AS already_linked, count(*) FILTER (WHERE current_link IS NULL AND candidates = 1) AS will_link, count(*) FILTER (WHERE current_link IS NULL AND candidates = 0) AS no_worker, count(*) FILTER (WHERE current_link IS NULL AND candidates > 1) AS ambiguous FROM m; -- The guard. `worker_profiles_org_email_key` should make this impossible; if it -- ever returns a row the backfill must not run, because picking one of two -- people is exactly the kind of quiet wrong answer this script exists to avoid. \echo '' \echo '── Ambiguous matches (must be empty) ──────────────────────────────────' SELECT a.id AS application_id, a.email, count(w.id) AS matching_workers FROM job_applications a JOIN worker_profiles w ON w.org_id = a.org_id AND lower(w.email::text) = lower(a.email::text) WHERE a.worker_profile_id IS NULL GROUP BY a.id, a.email HAVING count(w.id) > 1; \echo '' \echo '── Linking ────────────────────────────────────────────────────────────' UPDATE job_applications a SET worker_profile_id = w.id FROM worker_profiles w WHERE a.worker_profile_id IS NULL AND w.org_id = a.org_id AND lower(w.email::text) = lower(a.email::text) -- Belt and braces: only where exactly one profile matches. AND (SELECT count(*) FROM worker_profiles w2 WHERE w2.org_id = a.org_id AND lower(w2.email::text) = lower(a.email::text)) = 1; \echo '' \echo '── After ──────────────────────────────────────────────────────────────' WITH m AS ( SELECT a.worker_profile_id AS current_link, (SELECT count(*) FROM worker_profiles w WHERE w.org_id = a.org_id AND lower(w.email::text) = lower(a.email::text)) AS candidates FROM job_applications a ) SELECT count(*) AS applications, count(*) FILTER (WHERE current_link IS NOT NULL) AS linked, count(*) FILTER (WHERE current_link IS NULL AND candidates = 1) AS still_matchable, count(*) FILTER (WHERE current_link IS NULL AND candidates = 0) AS no_worker FROM m; COMMIT;