115 lines
5.1 KiB
PL/PgSQL
115 lines
5.1 KiB
PL/PgSQL
-- ============================================================================
|
|
-- One-off backfill: job_applications.worker_profile_id
|
|
--
|
|
-- NOT A MIGRATION, and deliberately not one. `worker_profile_id` has existed
|
|
-- since migration 000001 and the column is unchanged; what was missing was the
|
|
-- frontend setting it when it filed an application from the talent pool. That
|
|
-- is fixed going forward. This connects the rows written before the fix.
|
|
--
|
|
-- A migration would make this part of the schema's history and run once per
|
|
-- environment whether or not it was wanted. A script runs when somebody decides
|
|
-- to run it, against the environment they name, after reading what it will do.
|
|
--
|
|
-- WHAT IT MATCHES
|
|
--
|
|
-- org_id AND case-insensitive email
|
|
--
|
|
-- and nothing else. No fuzzy matching, no name comparison, no "probably the
|
|
-- same person". Email is the identity `worker_profiles` itself already asserts:
|
|
--
|
|
-- CREATE UNIQUE INDEX worker_profiles_org_email_key
|
|
-- ON worker_profiles (org_id, email);
|
|
--
|
|
-- That constraint is why an ambiguous match is not merely unlikely but
|
|
-- IMPOSSIBLE: at most one worker profile can exist for any (org_id, email), so
|
|
-- the join below can never find two. The HAVING clause asserts it anyway —
|
|
-- cheap, and it fails loudly rather than silently picking one if that
|
|
-- constraint is ever relaxed.
|
|
--
|
|
-- `email` is `citext`, so the comparison is already case-insensitive; it is
|
|
-- written with lower() so the intent survives a column type change.
|
|
--
|
|
-- WHAT IT TOUCHES
|
|
--
|
|
-- job_applications.worker_profile_id ONLY, and only where it is NULL.
|
|
--
|
|
-- No status, no score, no timestamp, no other column, no other table. It never
|
|
-- creates a worker_profiles row — an application with nobody behind it stays
|
|
-- unlinked, which is the honest answer.
|
|
--
|
|
-- SAFE TO RE-RUN. Rows already linked are excluded, so a second run changes
|
|
-- nothing.
|
|
--
|
|
-- USAGE
|
|
-- psql -d <database> -f scripts/backfill_worker_profile_id.sql
|
|
--
|
|
-- Wrapped in a transaction: the report and the update see the same rows, and a
|
|
-- failure leaves nothing half-applied.
|
|
-- ============================================================================
|
|
|
|
BEGIN;
|
|
|
|
\echo ''
|
|
\echo '── Before ─────────────────────────────────────────────────────────────'
|
|
|
|
WITH m AS (
|
|
SELECT a.worker_profile_id AS current_link,
|
|
(SELECT count(*) FROM worker_profiles w
|
|
WHERE w.org_id = a.org_id
|
|
AND lower(w.email::text) = lower(a.email::text)) AS candidates
|
|
FROM job_applications a
|
|
)
|
|
SELECT count(*) AS applications,
|
|
count(*) FILTER (WHERE current_link IS NOT NULL) AS already_linked,
|
|
count(*) FILTER (WHERE current_link IS NULL AND candidates = 1) AS will_link,
|
|
count(*) FILTER (WHERE current_link IS NULL AND candidates = 0) AS no_worker,
|
|
count(*) FILTER (WHERE current_link IS NULL AND candidates > 1) AS ambiguous
|
|
FROM m;
|
|
|
|
-- The guard. `worker_profiles_org_email_key` should make this impossible; if it
|
|
-- ever returns a row the backfill must not run, because picking one of two
|
|
-- people is exactly the kind of quiet wrong answer this script exists to avoid.
|
|
\echo ''
|
|
\echo '── Ambiguous matches (must be empty) ──────────────────────────────────'
|
|
|
|
SELECT a.id AS application_id, a.email, count(w.id) AS matching_workers
|
|
FROM job_applications a
|
|
JOIN worker_profiles w
|
|
ON w.org_id = a.org_id
|
|
AND lower(w.email::text) = lower(a.email::text)
|
|
WHERE a.worker_profile_id IS NULL
|
|
GROUP BY a.id, a.email
|
|
HAVING count(w.id) > 1;
|
|
|
|
\echo ''
|
|
\echo '── Linking ────────────────────────────────────────────────────────────'
|
|
|
|
UPDATE job_applications a
|
|
SET worker_profile_id = w.id
|
|
FROM worker_profiles w
|
|
WHERE a.worker_profile_id IS NULL
|
|
AND w.org_id = a.org_id
|
|
AND lower(w.email::text) = lower(a.email::text)
|
|
-- Belt and braces: only where exactly one profile matches.
|
|
AND (SELECT count(*) FROM worker_profiles w2
|
|
WHERE w2.org_id = a.org_id
|
|
AND lower(w2.email::text) = lower(a.email::text)) = 1;
|
|
|
|
\echo ''
|
|
\echo '── After ──────────────────────────────────────────────────────────────'
|
|
|
|
WITH m AS (
|
|
SELECT a.worker_profile_id AS current_link,
|
|
(SELECT count(*) FROM worker_profiles w
|
|
WHERE w.org_id = a.org_id
|
|
AND lower(w.email::text) = lower(a.email::text)) AS candidates
|
|
FROM job_applications a
|
|
)
|
|
SELECT count(*) AS applications,
|
|
count(*) FILTER (WHERE current_link IS NOT NULL) AS linked,
|
|
count(*) FILTER (WHERE current_link IS NULL AND candidates = 1) AS still_matchable,
|
|
count(*) FILTER (WHERE current_link IS NULL AND candidates = 0) AS no_worker
|
|
FROM m;
|
|
|
|
COMMIT;
|