301 lines
10 KiB
Go
301 lines
10 KiB
Go
package seeder_test
|
|
|
|
import (
|
|
"context"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/seeder"
|
|
"github.com/krow/krow-backend/go-api/internal/testutil"
|
|
)
|
|
|
|
// tableFor maps the fixture's entity names onto their tables, so the counts
|
|
// asserted below come from the frontend's own data rather than from literals.
|
|
var tableFor = map[string]string{
|
|
"JobPosting": "job_postings", "JobApplication": "job_applications",
|
|
"AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles",
|
|
"Course": "courses", "Badge": "badges", "LearningPath": "learning_paths",
|
|
"Certification": "certifications", "RoleCategory": "role_categories",
|
|
"UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments",
|
|
}
|
|
|
|
func count(t *testing.T, h *testutil.Harness, table string) int {
|
|
t.Helper()
|
|
var n int
|
|
if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil {
|
|
t.Fatalf("count %s: %v", table, err)
|
|
}
|
|
return n
|
|
}
|
|
|
|
// TestSeedMatchesFixtureCounts checks every entity against the fixture rather
|
|
// than against a hardcoded headline number.
|
|
func TestSeedMatchesFixtureCounts(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
|
|
for entity, table := range tableFor {
|
|
want := len(fx.Entities[entity])
|
|
if got := count(t, h, table); got != want {
|
|
t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestSeedRegressionAnchors pins the figures the demo dataset is built to
|
|
// produce. These are verified against the source, not assumed: the prompt's
|
|
// "6 postings / 22 applications" is 6 *active* postings and 24 applications.
|
|
func TestSeedRegressionAnchors(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
var active int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if active != 6 {
|
|
t.Errorf("active postings = %d, want 6", active)
|
|
}
|
|
if total := count(t, h, "job_postings"); total != 8 {
|
|
t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total)
|
|
}
|
|
if total := count(t, h, "job_applications"); total != 24 {
|
|
t.Errorf("applications = %d, want 24", total)
|
|
}
|
|
|
|
var scored int
|
|
var avgScored float64
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0").
|
|
Scan(&scored, &avgScored); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if scored != 9 {
|
|
t.Errorf("scored applications = %d, want 9", scored)
|
|
}
|
|
if avgScored < 75.95 || avgScored > 76.05 {
|
|
t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored)
|
|
}
|
|
|
|
var hires int
|
|
var avgHire float64
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if hires != 3 {
|
|
t.Errorf("hires = %d, want 3", hires)
|
|
}
|
|
if avgHire < 94.0 || avgHire > 94.5 {
|
|
t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire)
|
|
}
|
|
}
|
|
|
|
// TestSeedPreservesSourceValues compares stored rows field-by-field against the
|
|
// fixture, rather than trusting that the counts lining up means the data did.
|
|
func TestSeedPreservesSourceValues(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
ctx := context.Background()
|
|
|
|
for _, want := range fx.Entities["JobPosting"] {
|
|
legacy := want["id"].(string)
|
|
var title, status, company, roleCategory, createdDate string
|
|
var payMin, payMax int
|
|
err := h.Pool.QueryRow(ctx, `
|
|
SELECT title, status::text, company, role_category, pay_range_min, pay_range_max,
|
|
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
|
FROM job_postings WHERE legacy_id = $1`, legacy).
|
|
Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate)
|
|
if err != nil {
|
|
t.Fatalf("%s: %v", legacy, err)
|
|
}
|
|
if title != want["title"] {
|
|
t.Errorf("%s title = %q, want %q", legacy, title, want["title"])
|
|
}
|
|
if status != want["status"] {
|
|
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
|
}
|
|
if createdDate != want["created_date"] {
|
|
t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"])
|
|
}
|
|
if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) {
|
|
t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v))
|
|
}
|
|
if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) {
|
|
t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v))
|
|
}
|
|
}
|
|
|
|
// Applications carry updated_date in the source, and the gap from
|
|
// created_date is what buildHires reads as time-to-hire.
|
|
for _, want := range fx.Entities["JobApplication"] {
|
|
legacy := want["id"].(string)
|
|
var name, email, status, created, updated string
|
|
var score int
|
|
err := h.Pool.QueryRow(ctx, `
|
|
SELECT applicant_name, email::text, status::text, ai_score,
|
|
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'),
|
|
to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
|
FROM job_applications WHERE legacy_id = $1`, legacy).
|
|
Scan(&name, &email, &status, &score, &created, &updated)
|
|
if err != nil {
|
|
t.Fatalf("%s: %v", legacy, err)
|
|
}
|
|
if name != want["applicant_name"] {
|
|
t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"])
|
|
}
|
|
if status != want["status"] {
|
|
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
|
}
|
|
if created != want["created_date"] {
|
|
t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"])
|
|
}
|
|
if updated != want["updated_date"] {
|
|
t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"])
|
|
}
|
|
if v, ok := want["ai_score"].(float64); ok && score != int(v) {
|
|
t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v))
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestSeedIsIdempotent runs the seeder a second time over an already-seeded
|
|
// database and expects every count and every id to be unchanged.
|
|
func TestSeedIsIdempotent(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
ctx := context.Background()
|
|
|
|
before := map[string]int{}
|
|
for _, table := range tableFor {
|
|
before[table] = count(t, h, table)
|
|
}
|
|
var idsBefore string
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
|
Scan(&idsBefore); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil {
|
|
t.Fatalf("second seed: %v", err)
|
|
}
|
|
|
|
for _, table := range tableFor {
|
|
if got := count(t, h, table); got != before[table] {
|
|
t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows",
|
|
table, got, before[table])
|
|
}
|
|
}
|
|
var idsAfter string
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
|
Scan(&idsAfter); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if idsAfter != idsBefore {
|
|
t.Error("application ids changed across a re-seed; keys are not deterministic")
|
|
}
|
|
}
|
|
|
|
// TestSeedRelationships checks that every reference was rewritten to a real row.
|
|
func TestSeedRelationships(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
dangling := []struct{ name, query string }{
|
|
{"applications without a posting",
|
|
`SELECT count(*) FROM job_applications a
|
|
LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`},
|
|
{"interviews without an application",
|
|
`SELECT count(*) FROM ai_interviews i
|
|
LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`},
|
|
{"staff without an application",
|
|
`SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id
|
|
WHERE s.application_id IS NOT NULL AND a.id IS NULL`},
|
|
{"shifts without staff",
|
|
`SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id
|
|
WHERE r.staff_id IS NOT NULL AND s.id IS NULL`},
|
|
{"evidence without a course",
|
|
`SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id
|
|
WHERE e.course_id IS NOT NULL AND c.id IS NULL`},
|
|
}
|
|
for _, d := range dangling {
|
|
var n int
|
|
if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil {
|
|
t.Fatalf("%s: %v", d.name, err)
|
|
}
|
|
if n != 0 {
|
|
t.Errorf("%s: %d", d.name, n)
|
|
}
|
|
}
|
|
|
|
// interview_id is a soft reference on purpose: the source contains one
|
|
// dangling value (app_devon -> int_devon), and preserving it is the point.
|
|
var set, resolve int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL),
|
|
(SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`).
|
|
Scan(&set, &resolve); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if set != 5 {
|
|
t.Errorf("applications carrying interview_id = %d, want 5", set)
|
|
}
|
|
if resolve != 4 {
|
|
t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve)
|
|
}
|
|
}
|
|
|
|
// TestSeedOrganizationScope checks every seeded row belongs to the development
|
|
// organization, so organization scoping has something real to filter on.
|
|
func TestSeedOrganizationScope(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
for _, table := range tableFor {
|
|
var wrong int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID).
|
|
Scan(&wrong); err != nil {
|
|
t.Fatalf("%s: %v", table, err)
|
|
}
|
|
if wrong != 0 {
|
|
t.Errorf("%s: %d rows outside the development organization", table, wrong)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestDeterministicUUID pins the key derivation: the same source id must always
|
|
// produce the same key, or a re-seed would duplicate every row.
|
|
func TestDeterministicUUID(t *testing.T) {
|
|
a := seeder.DeterministicUUID("JobPosting:job_chef")
|
|
b := seeder.DeterministicUUID("JobPosting:job_chef")
|
|
if a != b {
|
|
t.Fatalf("not deterministic: %s != %s", a, b)
|
|
}
|
|
if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a {
|
|
t.Fatal("distinct source ids produced the same key")
|
|
}
|
|
if len(a) != 36 || a[14] != '5' {
|
|
t.Errorf("expected a v5 UUID, got %q", a)
|
|
}
|
|
}
|
|
|
|
// TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must
|
|
// produce identical shift records.
|
|
func TestSeedShiftsStableForAnchor(t *testing.T) {
|
|
anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local)
|
|
a := seeder.BuildShifts(anchor)
|
|
b := seeder.BuildShifts(anchor)
|
|
if len(a) != len(b) {
|
|
t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b))
|
|
}
|
|
for i := range a {
|
|
if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] {
|
|
t.Fatalf("shift %d differs between runs", i)
|
|
}
|
|
}
|
|
}
|