Files
krow_backend/go-api/internal/seeder/seeder_test.go
2026-08-24 13:06:29 +05:30

301 lines
10 KiB
Go

package seeder_test
import (
"context"
"testing"
"time"
"github.com/krow/krow-backend/go-api/internal/seeder"
"github.com/krow/krow-backend/go-api/internal/testutil"
)
// tableFor maps the fixture's entity names onto their tables, so the counts
// asserted below come from the frontend's own data rather than from literals.
var tableFor = map[string]string{
"JobPosting": "job_postings", "JobApplication": "job_applications",
"AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles",
"Course": "courses", "Badge": "badges", "LearningPath": "learning_paths",
"Certification": "certifications", "RoleCategory": "role_categories",
"UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments",
}
func count(t *testing.T, h *testutil.Harness, table string) int {
t.Helper()
var n int
if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil {
t.Fatalf("count %s: %v", table, err)
}
return n
}
// TestSeedMatchesFixtureCounts checks every entity against the fixture rather
// than against a hardcoded headline number.
func TestSeedMatchesFixtureCounts(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
for entity, table := range tableFor {
want := len(fx.Entities[entity])
if got := count(t, h, table); got != want {
t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want)
}
}
}
// TestSeedRegressionAnchors pins the figures the demo dataset is built to
// produce. These are verified against the source, not assumed: the prompt's
// "6 postings / 22 applications" is 6 *active* postings and 24 applications.
func TestSeedRegressionAnchors(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
var active int
if err := h.Pool.QueryRow(ctx,
"SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil {
t.Fatal(err)
}
if active != 6 {
t.Errorf("active postings = %d, want 6", active)
}
if total := count(t, h, "job_postings"); total != 8 {
t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total)
}
if total := count(t, h, "job_applications"); total != 24 {
t.Errorf("applications = %d, want 24", total)
}
var scored int
var avgScored float64
if err := h.Pool.QueryRow(ctx,
"SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0").
Scan(&scored, &avgScored); err != nil {
t.Fatal(err)
}
if scored != 9 {
t.Errorf("scored applications = %d, want 9", scored)
}
if avgScored < 75.95 || avgScored > 76.05 {
t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored)
}
var hires int
var avgHire float64
if err := h.Pool.QueryRow(ctx,
"SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil {
t.Fatal(err)
}
if hires != 3 {
t.Errorf("hires = %d, want 3", hires)
}
if avgHire < 94.0 || avgHire > 94.5 {
t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire)
}
}
// TestSeedPreservesSourceValues compares stored rows field-by-field against the
// fixture, rather than trusting that the counts lining up means the data did.
func TestSeedPreservesSourceValues(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
for _, want := range fx.Entities["JobPosting"] {
legacy := want["id"].(string)
var title, status, company, roleCategory, createdDate string
var payMin, payMax int
err := h.Pool.QueryRow(ctx, `
SELECT title, status::text, company, role_category, pay_range_min, pay_range_max,
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
FROM job_postings WHERE legacy_id = $1`, legacy).
Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate)
if err != nil {
t.Fatalf("%s: %v", legacy, err)
}
if title != want["title"] {
t.Errorf("%s title = %q, want %q", legacy, title, want["title"])
}
if status != want["status"] {
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
}
if createdDate != want["created_date"] {
t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"])
}
if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) {
t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v))
}
if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) {
t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v))
}
}
// Applications carry updated_date in the source, and the gap from
// created_date is what buildHires reads as time-to-hire.
for _, want := range fx.Entities["JobApplication"] {
legacy := want["id"].(string)
var name, email, status, created, updated string
var score int
err := h.Pool.QueryRow(ctx, `
SELECT applicant_name, email::text, status::text, ai_score,
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'),
to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
FROM job_applications WHERE legacy_id = $1`, legacy).
Scan(&name, &email, &status, &score, &created, &updated)
if err != nil {
t.Fatalf("%s: %v", legacy, err)
}
if name != want["applicant_name"] {
t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"])
}
if status != want["status"] {
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
}
if created != want["created_date"] {
t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"])
}
if updated != want["updated_date"] {
t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"])
}
if v, ok := want["ai_score"].(float64); ok && score != int(v) {
t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v))
}
}
}
// TestSeedIsIdempotent runs the seeder a second time over an already-seeded
// database and expects every count and every id to be unchanged.
func TestSeedIsIdempotent(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
before := map[string]int{}
for _, table := range tableFor {
before[table] = count(t, h, table)
}
var idsBefore string
if err := h.Pool.QueryRow(ctx,
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
Scan(&idsBefore); err != nil {
t.Fatal(err)
}
if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil {
t.Fatalf("second seed: %v", err)
}
for _, table := range tableFor {
if got := count(t, h, table); got != before[table] {
t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows",
table, got, before[table])
}
}
var idsAfter string
if err := h.Pool.QueryRow(ctx,
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
Scan(&idsAfter); err != nil {
t.Fatal(err)
}
if idsAfter != idsBefore {
t.Error("application ids changed across a re-seed; keys are not deterministic")
}
}
// TestSeedRelationships checks that every reference was rewritten to a real row.
func TestSeedRelationships(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
dangling := []struct{ name, query string }{
{"applications without a posting",
`SELECT count(*) FROM job_applications a
LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`},
{"interviews without an application",
`SELECT count(*) FROM ai_interviews i
LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`},
{"staff without an application",
`SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id
WHERE s.application_id IS NOT NULL AND a.id IS NULL`},
{"shifts without staff",
`SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id
WHERE r.staff_id IS NOT NULL AND s.id IS NULL`},
{"evidence without a course",
`SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id
WHERE e.course_id IS NOT NULL AND c.id IS NULL`},
}
for _, d := range dangling {
var n int
if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil {
t.Fatalf("%s: %v", d.name, err)
}
if n != 0 {
t.Errorf("%s: %d", d.name, n)
}
}
// interview_id is a soft reference on purpose: the source contains one
// dangling value (app_devon -> int_devon), and preserving it is the point.
var set, resolve int
if err := h.Pool.QueryRow(ctx,
`SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL),
(SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`).
Scan(&set, &resolve); err != nil {
t.Fatal(err)
}
if set != 5 {
t.Errorf("applications carrying interview_id = %d, want 5", set)
}
if resolve != 4 {
t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve)
}
}
// TestSeedOrganizationScope checks every seeded row belongs to the development
// organization, so organization scoping has something real to filter on.
func TestSeedOrganizationScope(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
for _, table := range tableFor {
var wrong int
if err := h.Pool.QueryRow(ctx,
"SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID).
Scan(&wrong); err != nil {
t.Fatalf("%s: %v", table, err)
}
if wrong != 0 {
t.Errorf("%s: %d rows outside the development organization", table, wrong)
}
}
}
// TestDeterministicUUID pins the key derivation: the same source id must always
// produce the same key, or a re-seed would duplicate every row.
func TestDeterministicUUID(t *testing.T) {
a := seeder.DeterministicUUID("JobPosting:job_chef")
b := seeder.DeterministicUUID("JobPosting:job_chef")
if a != b {
t.Fatalf("not deterministic: %s != %s", a, b)
}
if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a {
t.Fatal("distinct source ids produced the same key")
}
if len(a) != 36 || a[14] != '5' {
t.Errorf("expected a v5 UUID, got %q", a)
}
}
// TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must
// produce identical shift records.
func TestSeedShiftsStableForAnchor(t *testing.T) {
anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local)
a := seeder.BuildShifts(anchor)
b := seeder.BuildShifts(anchor)
if len(a) != len(b) {
t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b))
}
for i := range a {
if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] {
t.Fatalf("shift %d differs between runs", i)
}
}
}