417 lines
15 KiB
Go
417 lines
15 KiB
Go
package seeder_test
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/seeder"
|
|
"github.com/krow/krow-backend/go-api/internal/testutil"
|
|
)
|
|
|
|
// tableFor maps the fixture's entity names onto their tables, so the counts
|
|
// asserted below come from the frontend's own data rather than from literals.
|
|
var tableFor = map[string]string{
|
|
"JobPosting": "job_postings", "JobApplication": "job_applications",
|
|
"AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles",
|
|
"Course": "courses", "Badge": "badges", "LearningPath": "learning_paths",
|
|
"Certification": "certifications", "RoleCategory": "role_categories",
|
|
"UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments",
|
|
}
|
|
|
|
func count(t *testing.T, h *testutil.Harness, table string) int {
|
|
t.Helper()
|
|
var n int
|
|
if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil {
|
|
t.Fatalf("count %s: %v", table, err)
|
|
}
|
|
return n
|
|
}
|
|
|
|
// TestSeedMatchesFixtureCounts checks every entity against the fixture rather
|
|
// than against a hardcoded headline number.
|
|
func TestSeedMatchesFixtureCounts(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
|
|
for entity, table := range tableFor {
|
|
want := len(fx.Entities[entity])
|
|
if got := count(t, h, table); got != want {
|
|
t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestSeedRegressionAnchors pins the figures the demo dataset is built to
|
|
// produce. These are verified against the source, not assumed: the prompt's
|
|
// "6 postings / 22 applications" is 6 *active* postings and 24 applications.
|
|
func TestSeedRegressionAnchors(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
var active int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if active != 6 {
|
|
t.Errorf("active postings = %d, want 6", active)
|
|
}
|
|
if total := count(t, h, "job_postings"); total != 8 {
|
|
t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total)
|
|
}
|
|
if total := count(t, h, "job_applications"); total != 24 {
|
|
t.Errorf("applications = %d, want 24", total)
|
|
}
|
|
|
|
var scored int
|
|
var avgScored float64
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0").
|
|
Scan(&scored, &avgScored); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if scored != 9 {
|
|
t.Errorf("scored applications = %d, want 9", scored)
|
|
}
|
|
if avgScored < 75.95 || avgScored > 76.05 {
|
|
t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored)
|
|
}
|
|
|
|
var hires int
|
|
var avgHire float64
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if hires != 3 {
|
|
t.Errorf("hires = %d, want 3", hires)
|
|
}
|
|
if avgHire < 94.0 || avgHire > 94.5 {
|
|
t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire)
|
|
}
|
|
}
|
|
|
|
// TestSeedPreservesSourceValues compares stored rows field-by-field against the
|
|
// fixture, rather than trusting that the counts lining up means the data did.
|
|
func TestSeedPreservesSourceValues(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
ctx := context.Background()
|
|
|
|
for _, want := range fx.Entities["JobPosting"] {
|
|
legacy := want["id"].(string)
|
|
var title, status, company, roleCategory, createdDate string
|
|
var payMin, payMax int
|
|
err := h.Pool.QueryRow(ctx, `
|
|
SELECT title, status::text, company, role_category, pay_range_min, pay_range_max,
|
|
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
|
FROM job_postings WHERE legacy_id = $1`, legacy).
|
|
Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate)
|
|
if err != nil {
|
|
t.Fatalf("%s: %v", legacy, err)
|
|
}
|
|
if title != want["title"] {
|
|
t.Errorf("%s title = %q, want %q", legacy, title, want["title"])
|
|
}
|
|
if status != want["status"] {
|
|
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
|
}
|
|
if createdDate != want["created_date"] {
|
|
t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"])
|
|
}
|
|
if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) {
|
|
t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v))
|
|
}
|
|
if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) {
|
|
t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v))
|
|
}
|
|
}
|
|
|
|
// Applications carry updated_date in the source, and the gap from
|
|
// created_date is what buildHires reads as time-to-hire.
|
|
for _, want := range fx.Entities["JobApplication"] {
|
|
legacy := want["id"].(string)
|
|
var name, email, status, created, updated string
|
|
var score int
|
|
err := h.Pool.QueryRow(ctx, `
|
|
SELECT applicant_name, email::text, status::text, ai_score,
|
|
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'),
|
|
to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
|
FROM job_applications WHERE legacy_id = $1`, legacy).
|
|
Scan(&name, &email, &status, &score, &created, &updated)
|
|
if err != nil {
|
|
t.Fatalf("%s: %v", legacy, err)
|
|
}
|
|
if name != want["applicant_name"] {
|
|
t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"])
|
|
}
|
|
if status != want["status"] {
|
|
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
|
}
|
|
if created != want["created_date"] {
|
|
t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"])
|
|
}
|
|
if updated != want["updated_date"] {
|
|
t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"])
|
|
}
|
|
if v, ok := want["ai_score"].(float64); ok && score != int(v) {
|
|
t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v))
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestSeedIsIdempotent runs the seeder a second time over an already-seeded
|
|
// database and expects every count and every id to be unchanged.
|
|
func TestSeedIsIdempotent(t *testing.T) {
|
|
h := testutil.New(t)
|
|
fx := testutil.Fixture(t)
|
|
ctx := context.Background()
|
|
|
|
before := map[string]int{}
|
|
for _, table := range tableFor {
|
|
before[table] = count(t, h, table)
|
|
}
|
|
var idsBefore string
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
|
Scan(&idsBefore); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil {
|
|
t.Fatalf("second seed: %v", err)
|
|
}
|
|
|
|
for _, table := range tableFor {
|
|
if got := count(t, h, table); got != before[table] {
|
|
t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows",
|
|
table, got, before[table])
|
|
}
|
|
}
|
|
var idsAfter string
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
|
Scan(&idsAfter); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if idsAfter != idsBefore {
|
|
t.Error("application ids changed across a re-seed; keys are not deterministic")
|
|
}
|
|
}
|
|
|
|
// TestSeedRelationships checks that every reference was rewritten to a real row.
|
|
func TestSeedRelationships(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
dangling := []struct{ name, query string }{
|
|
{"applications without a posting",
|
|
`SELECT count(*) FROM job_applications a
|
|
LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`},
|
|
{"interviews without an application",
|
|
`SELECT count(*) FROM ai_interviews i
|
|
LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`},
|
|
{"staff without an application",
|
|
`SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id
|
|
WHERE s.application_id IS NOT NULL AND a.id IS NULL`},
|
|
{"shifts without staff",
|
|
`SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id
|
|
WHERE r.staff_id IS NOT NULL AND s.id IS NULL`},
|
|
{"evidence without a course",
|
|
`SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id
|
|
WHERE e.course_id IS NOT NULL AND c.id IS NULL`},
|
|
}
|
|
for _, d := range dangling {
|
|
var n int
|
|
if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil {
|
|
t.Fatalf("%s: %v", d.name, err)
|
|
}
|
|
if n != 0 {
|
|
t.Errorf("%s: %d", d.name, n)
|
|
}
|
|
}
|
|
|
|
// interview_id is a soft reference on purpose: the source contains one
|
|
// dangling value (app_devon -> int_devon), and preserving it is the point.
|
|
var set, resolve int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL),
|
|
(SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`).
|
|
Scan(&set, &resolve); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if set != 5 {
|
|
t.Errorf("applications carrying interview_id = %d, want 5", set)
|
|
}
|
|
if resolve != 4 {
|
|
t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve)
|
|
}
|
|
}
|
|
|
|
// TestSeedOrganizationScope checks every seeded row belongs to the development
|
|
// organization, so organization scoping has something real to filter on.
|
|
func TestSeedOrganizationScope(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
for _, table := range tableFor {
|
|
var wrong int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
"SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID).
|
|
Scan(&wrong); err != nil {
|
|
t.Fatalf("%s: %v", table, err)
|
|
}
|
|
if wrong != 0 {
|
|
t.Errorf("%s: %d rows outside the development organization", table, wrong)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestDeterministicUUID pins the key derivation: the same source id must always
|
|
// produce the same key, or a re-seed would duplicate every row.
|
|
func TestDeterministicUUID(t *testing.T) {
|
|
a := seeder.DeterministicUUID("JobPosting:job_chef")
|
|
b := seeder.DeterministicUUID("JobPosting:job_chef")
|
|
if a != b {
|
|
t.Fatalf("not deterministic: %s != %s", a, b)
|
|
}
|
|
if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a {
|
|
t.Fatal("distinct source ids produced the same key")
|
|
}
|
|
if len(a) != 36 || a[14] != '5' {
|
|
t.Errorf("expected a v5 UUID, got %q", a)
|
|
}
|
|
}
|
|
|
|
// TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must
|
|
// produce identical shift records.
|
|
func TestSeedShiftsStableForAnchor(t *testing.T) {
|
|
anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local)
|
|
a := seeder.BuildShifts(anchor)
|
|
b := seeder.BuildShifts(anchor)
|
|
if len(a) != len(b) {
|
|
t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b))
|
|
}
|
|
for i := range a {
|
|
if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] {
|
|
t.Fatalf("shift %d differs between runs", i)
|
|
}
|
|
}
|
|
}
|
|
|
|
// The seed must exercise every application_status. It produced four of seven —
|
|
// `shortlisted`, `rejected` and `assigned` existed only in the schema — and that
|
|
// gap was hiding real defects rather than merely being incomplete: the funnel
|
|
// dropped `rejected` and `assigned` out of every stage bucket, and the agent's
|
|
// candidate lookup offered somebody already on a shift as a person to chase.
|
|
// Neither is reachable by any test written against data that never produces them.
|
|
func TestSeedExercisesEveryApplicationStatus(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
rows, err := h.Pool.Query(ctx, `SELECT unnest(enum_range(NULL::application_status))::text`)
|
|
if err != nil {
|
|
t.Fatalf("read enum: %v", err)
|
|
}
|
|
defer rows.Close()
|
|
var declared []string
|
|
for rows.Next() {
|
|
var s string
|
|
if err := rows.Scan(&s); err != nil {
|
|
t.Fatalf("scan enum: %v", err)
|
|
}
|
|
declared = append(declared, s)
|
|
}
|
|
if err := rows.Err(); err != nil {
|
|
t.Fatalf("enum rows: %v", err)
|
|
}
|
|
if len(declared) == 0 {
|
|
t.Fatal("application_status has no values")
|
|
}
|
|
|
|
for _, status := range declared {
|
|
var n int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT count(*) FROM job_applications WHERE status = $1::application_status`,
|
|
status).Scan(&n); err != nil {
|
|
t.Fatalf("count %s: %v", status, err)
|
|
}
|
|
if n == 0 {
|
|
t.Errorf("no seeded application is %q — nothing can test the paths that handle it", status)
|
|
}
|
|
}
|
|
}
|
|
|
|
// `assigned` is only meaningful if an assignment row backs it: the status says
|
|
// somebody is on a shift, and without the row it says it of nobody.
|
|
func TestAnAssignedApplicationHasAnAssignment(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
|
|
var orphans int
|
|
if err := h.Pool.QueryRow(ctx, `
|
|
SELECT count(*) FROM job_applications a
|
|
WHERE a.status = 'assigned'
|
|
AND NOT EXISTS (
|
|
SELECT 1 FROM assignments s
|
|
WHERE s.application_id = a.id AND s.org_id = a.org_id)`).Scan(&orphans); err != nil {
|
|
t.Fatalf("count orphans: %v", err)
|
|
}
|
|
if orphans != 0 {
|
|
t.Errorf("%d applications are 'assigned' with no assignment row behind them", orphans)
|
|
}
|
|
|
|
// And the reverse: every position still reports the empty case honestly.
|
|
var positions, withAssignment int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT count(*), count(*) FILTER (WHERE EXISTS (
|
|
SELECT 1 FROM assignments s WHERE s.job_posting_id = p.id))
|
|
FROM job_postings p`).Scan(&positions, &withAssignment); err != nil {
|
|
t.Fatalf("count positions: %v", err)
|
|
}
|
|
if withAssignment == 0 || withAssignment >= positions {
|
|
t.Errorf("%d of %d positions have an assignment; want some but not all, so both "+
|
|
"the populated and the empty case are covered", withAssignment, positions)
|
|
}
|
|
}
|
|
|
|
// seed.json is generated from krow-demo/src/api/seed.js. The two used to be
|
|
// hand-maintained copies of one dataset, which fails quietly: the demo and the
|
|
// API answer the same question with different numbers, and the first symptom is
|
|
// a page disagreeing with an agent.
|
|
//
|
|
// This asserts the fixture is a *generated artefact*, not that it is *current*.
|
|
// Currency needs the generator, which is JavaScript and lives in the other
|
|
// repository — the frontend suite runs it and compares byte-for-byte, and
|
|
// `make seed-fixture-check` runs the same comparison from here. What this
|
|
// catches is a fixture built or replaced by hand, which carries no marker and
|
|
// would otherwise be indistinguishable from a generated one.
|
|
func TestFixtureIsGeneratedNotHandWritten(t *testing.T) {
|
|
path := os.Getenv("SEED_FIXTURE_PATH")
|
|
if path == "" {
|
|
path = filepath.Join("..", "..", "..", "seed", "fixtures", "seed.json")
|
|
}
|
|
raw, err := os.ReadFile(path)
|
|
if err != nil {
|
|
t.Skipf("fixture not readable at %s: %v", path, err)
|
|
}
|
|
|
|
var head struct {
|
|
Generated string `json:"_generated"`
|
|
}
|
|
if err := json.Unmarshal(raw, &head); err != nil {
|
|
t.Fatalf("fixture is not valid JSON: %v", err)
|
|
}
|
|
if head.Generated == "" {
|
|
t.Error("seed.json carries no `_generated` marker — it looks hand-written. " +
|
|
"Regenerate it with `make seed-fixture` rather than editing it directly.")
|
|
}
|
|
if !strings.Contains(head.Generated, "seed.js") {
|
|
t.Errorf("`_generated` does not name its source: %q", head.Generated)
|
|
}
|
|
}
|