package seeder_test import ( "context" "testing" "time" "github.com/krow/krow-backend/go-api/internal/seeder" "github.com/krow/krow-backend/go-api/internal/testutil" ) // tableFor maps the fixture's entity names onto their tables, so the counts // asserted below come from the frontend's own data rather than from literals. var tableFor = map[string]string{ "JobPosting": "job_postings", "JobApplication": "job_applications", "AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles", "Course": "courses", "Badge": "badges", "LearningPath": "learning_paths", "Certification": "certifications", "RoleCategory": "role_categories", "UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments", } func count(t *testing.T, h *testutil.Harness, table string) int { t.Helper() var n int if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil { t.Fatalf("count %s: %v", table, err) } return n } // TestSeedMatchesFixtureCounts checks every entity against the fixture rather // than against a hardcoded headline number. func TestSeedMatchesFixtureCounts(t *testing.T) { h := testutil.New(t) fx := testutil.Fixture(t) for entity, table := range tableFor { want := len(fx.Entities[entity]) if got := count(t, h, table); got != want { t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want) } } } // TestSeedRegressionAnchors pins the figures the demo dataset is built to // produce. These are verified against the source, not assumed: the prompt's // "6 postings / 22 applications" is 6 *active* postings and 24 applications. func TestSeedRegressionAnchors(t *testing.T) { h := testutil.New(t) ctx := context.Background() var active int if err := h.Pool.QueryRow(ctx, "SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil { t.Fatal(err) } if active != 6 { t.Errorf("active postings = %d, want 6", active) } if total := count(t, h, "job_postings"); total != 8 { t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total) } if total := count(t, h, "job_applications"); total != 24 { t.Errorf("applications = %d, want 24", total) } var scored int var avgScored float64 if err := h.Pool.QueryRow(ctx, "SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0"). Scan(&scored, &avgScored); err != nil { t.Fatal(err) } if scored != 9 { t.Errorf("scored applications = %d, want 9", scored) } if avgScored < 75.95 || avgScored > 76.05 { t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored) } var hires int var avgHire float64 if err := h.Pool.QueryRow(ctx, "SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil { t.Fatal(err) } if hires != 3 { t.Errorf("hires = %d, want 3", hires) } if avgHire < 94.0 || avgHire > 94.5 { t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire) } } // TestSeedPreservesSourceValues compares stored rows field-by-field against the // fixture, rather than trusting that the counts lining up means the data did. func TestSeedPreservesSourceValues(t *testing.T) { h := testutil.New(t) fx := testutil.Fixture(t) ctx := context.Background() for _, want := range fx.Entities["JobPosting"] { legacy := want["id"].(string) var title, status, company, roleCategory, createdDate string var payMin, payMax int err := h.Pool.QueryRow(ctx, ` SELECT title, status::text, company, role_category, pay_range_min, pay_range_max, to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"') FROM job_postings WHERE legacy_id = $1`, legacy). Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate) if err != nil { t.Fatalf("%s: %v", legacy, err) } if title != want["title"] { t.Errorf("%s title = %q, want %q", legacy, title, want["title"]) } if status != want["status"] { t.Errorf("%s status = %q, want %q", legacy, status, want["status"]) } if createdDate != want["created_date"] { t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"]) } if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) { t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v)) } if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) { t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v)) } } // Applications carry updated_date in the source, and the gap from // created_date is what buildHires reads as time-to-hire. for _, want := range fx.Entities["JobApplication"] { legacy := want["id"].(string) var name, email, status, created, updated string var score int err := h.Pool.QueryRow(ctx, ` SELECT applicant_name, email::text, status::text, ai_score, to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'), to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"') FROM job_applications WHERE legacy_id = $1`, legacy). Scan(&name, &email, &status, &score, &created, &updated) if err != nil { t.Fatalf("%s: %v", legacy, err) } if name != want["applicant_name"] { t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"]) } if status != want["status"] { t.Errorf("%s status = %q, want %q", legacy, status, want["status"]) } if created != want["created_date"] { t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"]) } if updated != want["updated_date"] { t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"]) } if v, ok := want["ai_score"].(float64); ok && score != int(v) { t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v)) } } } // TestSeedIsIdempotent runs the seeder a second time over an already-seeded // database and expects every count and every id to be unchanged. func TestSeedIsIdempotent(t *testing.T) { h := testutil.New(t) fx := testutil.Fixture(t) ctx := context.Background() before := map[string]int{} for _, table := range tableFor { before[table] = count(t, h, table) } var idsBefore string if err := h.Pool.QueryRow(ctx, "SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications"). Scan(&idsBefore); err != nil { t.Fatal(err) } if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil { t.Fatalf("second seed: %v", err) } for _, table := range tableFor { if got := count(t, h, table); got != before[table] { t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows", table, got, before[table]) } } var idsAfter string if err := h.Pool.QueryRow(ctx, "SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications"). Scan(&idsAfter); err != nil { t.Fatal(err) } if idsAfter != idsBefore { t.Error("application ids changed across a re-seed; keys are not deterministic") } } // TestSeedRelationships checks that every reference was rewritten to a real row. func TestSeedRelationships(t *testing.T) { h := testutil.New(t) ctx := context.Background() dangling := []struct{ name, query string }{ {"applications without a posting", `SELECT count(*) FROM job_applications a LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`}, {"interviews without an application", `SELECT count(*) FROM ai_interviews i LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`}, {"staff without an application", `SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id WHERE s.application_id IS NOT NULL AND a.id IS NULL`}, {"shifts without staff", `SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id WHERE r.staff_id IS NOT NULL AND s.id IS NULL`}, {"evidence without a course", `SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id WHERE e.course_id IS NOT NULL AND c.id IS NULL`}, } for _, d := range dangling { var n int if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil { t.Fatalf("%s: %v", d.name, err) } if n != 0 { t.Errorf("%s: %d", d.name, n) } } // interview_id is a soft reference on purpose: the source contains one // dangling value (app_devon -> int_devon), and preserving it is the point. var set, resolve int if err := h.Pool.QueryRow(ctx, `SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL), (SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`). Scan(&set, &resolve); err != nil { t.Fatal(err) } if set != 5 { t.Errorf("applications carrying interview_id = %d, want 5", set) } if resolve != 4 { t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve) } } // TestSeedOrganizationScope checks every seeded row belongs to the development // organization, so organization scoping has something real to filter on. func TestSeedOrganizationScope(t *testing.T) { h := testutil.New(t) ctx := context.Background() for _, table := range tableFor { var wrong int if err := h.Pool.QueryRow(ctx, "SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID). Scan(&wrong); err != nil { t.Fatalf("%s: %v", table, err) } if wrong != 0 { t.Errorf("%s: %d rows outside the development organization", table, wrong) } } } // TestDeterministicUUID pins the key derivation: the same source id must always // produce the same key, or a re-seed would duplicate every row. func TestDeterministicUUID(t *testing.T) { a := seeder.DeterministicUUID("JobPosting:job_chef") b := seeder.DeterministicUUID("JobPosting:job_chef") if a != b { t.Fatalf("not deterministic: %s != %s", a, b) } if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a { t.Fatal("distinct source ids produced the same key") } if len(a) != 36 || a[14] != '5' { t.Errorf("expected a v5 UUID, got %q", a) } } // TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must // produce identical shift records. func TestSeedShiftsStableForAnchor(t *testing.T) { anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local) a := seeder.BuildShifts(anchor) b := seeder.BuildShifts(anchor) if len(a) != len(b) { t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b)) } for i := range a { if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] { t.Fatalf("shift %d differs between runs", i) } } }