first commit

This commit is contained in:
2026-08-24 13:06:29 +05:30
commit 7d12ebef3d
86 changed files with 39996 additions and 0 deletions

View File

@@ -0,0 +1,525 @@
// Package seeder loads the frontend's demo dataset into PostgreSQL.
//
// Source of truth is the frontend, not this package. `seed/fixtures/seed.json`
// is produced by executing src/api/seed.js through Vite and serialising what it
// exports, so ids, dates, numbers and enum values arrive exactly as the demo
// has them — no transcription step, and nothing to drift. Shift records are the
// one exception: they are generated (see shifts.go) because their dates are
// anchored to now.
//
// Idempotency strategy: EXPLICIT UPSERT inside a single transaction.
//
// Every record's primary key is derived deterministically from its source id
// (uuid v5 over a fixed namespace), so re-running the seeder targets exactly the
// same rows and `ON CONFLICT (id) DO UPDATE` restores each one to its seeded
// values. Records created through the API survive a re-seed. A column the
// fixture does not carry is left as it is.
//
// Shift records are the one collection that is also PRUNED — see
// pruneShiftRecords. Upsert alone cannot converge a rolling window, and shift
// records are the only collection that is a rolling window.
package seeder
import (
"context"
"crypto/sha1"
"encoding/json"
"fmt"
"os"
"sort"
"strings"
"time"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgxpool"
"github.com/krow/krow-backend/go-api/internal/domain"
"github.com/krow/krow-backend/go-api/internal/orgctx"
)
// namespace is a fixed UUID used to derive record ids from source ids. Changing
// it re-keys the entire dataset, so it is a constant, not configuration.
var namespace = [16]byte{
0x6b, 0x72, 0x6f, 0x77, 0x2d, 0x73, 0x65, 0x65,
0x64, 0x2d, 0x76, 0x31, 0x00, 0x00, 0x00, 0x01,
}
// DeterministicUUID derives a stable v5 UUID from a source id.
func DeterministicUUID(name string) string {
h := sha1.New()
h.Write(namespace[:])
h.Write([]byte(name))
var b [16]byte
copy(b[:], h.Sum(nil))
b[6] = (b[6] & 0x0f) | 0x50 // version 5
b[8] = (b[8] & 0x3f) | 0x80 // RFC 4122 variant
return fmt.Sprintf("%x-%x-%x-%x-%x", b[0:4], b[4:6], b[6:8], b[8:10], b[10:16])
}
// Fixture is the serialised frontend dataset.
type Fixture struct {
DemoUser map[string]any `json:"demoUser"`
Entities map[string][]map[string]any `json:"entities"`
}
// Result counts what was written, by entity.
type Result struct {
OrgID string
Counts map[string]int
// Pruned is how many stale shift records this run removed. Reported rather
// than silent: a delete during a seed should never be something you have to
// read the source to discover.
Pruned int
}
// entityOrder is insertion order, chosen so every foreign key is satisfied by
// the time it is referenced.
var entityOrder = []string{
"RoleCategory", "Certification", "Badge", "Course", "LearningPath",
"JobPosting", "WorkerProfile", "JobApplication", "AIInterview", "Staff",
"Assignment", "ShiftRecord", "Evidence", "UserActivity",
}
// entityTable maps a frontend entity name to its table.
var entityTable = map[string]string{
"RoleCategory": "role_categories", "Certification": "certifications",
"Badge": "badges", "Course": "courses", "LearningPath": "learning_paths",
"JobPosting": "job_postings", "WorkerProfile": "worker_profiles",
"JobApplication": "job_applications", "AIInterview": "ai_interviews",
"Staff": "staff", "Assignment": "assignments", "ShiftRecord": "shift_records",
"Evidence": "evidence", "UserActivity": "user_activity",
}
// referenceFields are columns holding a source id that must be rewritten to the
// derived UUID. Every one of these is a real reference in the frontend data.
var referenceFields = map[string]bool{
"job_posting_id": true, "application_id": true, "course_id": true,
"staff_id": true, "assignment_id": true, "worker_profile_id": true,
"interview_id": true, "position_id": true, "candidate_id": true,
"user_id": true, "created_by": true,
}
// droppedFields are keys the fixture carries that no column exists for and no
// frontend code reads. Dropping them is deliberate and recorded here rather
// than being silent.
//
// _order — a positional index used only while seed.js builds its course list
// (src/api/seed.js:1326). Nothing reads it.
var droppedFields = map[string]bool{"_order": true}
// Seeder loads a fixture into a database.
type Seeder struct {
pool *pgxpool.Pool
fixture *Fixture
now time.Time
}
// Load reads a fixture from disk.
func Load(path string) (*Fixture, error) {
raw, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("read fixture %s: %w", path, err)
}
var f Fixture
if err := json.Unmarshal(raw, &f); err != nil {
return nil, fmt.Errorf("parse fixture %s: %w", path, err)
}
return &f, nil
}
// New builds a seeder. `now` anchors the generated shift records.
func New(pool *pgxpool.Pool, fixture *Fixture, now time.Time) *Seeder {
return &Seeder{pool: pool, fixture: fixture, now: now}
}
// Run seeds everything in one transaction: either the whole dataset lands or
// none of it does.
func (s *Seeder) Run(ctx context.Context) (*Result, error) {
tx, err := s.pool.Begin(ctx)
if err != nil {
return nil, err
}
defer func() { _ = tx.Rollback(ctx) }()
orgID, err := s.upsertOrganization(ctx, tx)
if err != nil {
return nil, err
}
result := &Result{OrgID: orgID, Counts: map[string]int{}}
n, err := s.upsertUser(ctx, tx, orgID)
if err != nil {
return nil, err
}
result.Counts["User"] = n
for _, entity := range entityOrder {
records := s.fixture.Entities[entity]
if entity == "ShiftRecord" {
records = BuildShifts(s.now)
}
count, err := s.upsertEntity(ctx, tx, orgID, entity, records)
if err != nil {
return nil, fmt.Errorf("seed %s: %w", entity, err)
}
result.Counts[entity] = count
if entity == "ShiftRecord" {
pruned, err := s.pruneShiftRecords(ctx, tx, orgID, records)
if err != nil {
return nil, fmt.Errorf("prune ShiftRecord: %w", err)
}
result.Pruned = pruned
}
}
if err := tx.Commit(ctx); err != nil {
return nil, err
}
return result, nil
}
// upsertOrganization creates the development organization the whole dataset
// belongs to. See internal/orgctx — this is not a tenant, it is a placeholder
// with a stable id so re-seeding is idempotent.
func (s *Seeder) upsertOrganization(ctx context.Context, tx pgx.Tx) (string, error) {
id := DeterministicUUID("org:" + orgctx.DevOrgSlug)
_, err := tx.Exec(ctx,
`INSERT INTO organizations (id, name, slug) VALUES ($1::uuid, $2, $3::citext)
ON CONFLICT (id) DO UPDATE SET name = EXCLUDED.name, updated_date = now()`,
id, orgctx.DevOrgName, orgctx.DevOrgSlug)
return id, err
}
// upsertUser writes the demo user and splits its preferences into their own
// table, as api-contract.md §9 describes.
func (s *Seeder) upsertUser(ctx context.Context, tx pgx.Tx, orgID string) (int, error) {
u := s.fixture.DemoUser
if u == nil {
return 0, nil
}
legacy, _ := u["id"].(string)
id := DeterministicUUID("User:" + legacy)
created := stringOr(u["created_date"], iso(s.now))
_, err := tx.Exec(ctx,
`INSERT INTO users (id, legacy_id, org_id, email, full_name, role, account_type, created_date, updated_date)
VALUES ($1::uuid, $2::text, $3::uuid, $4::citext, $5::text, $6::text, $7::text, $8::timestamptz, $8::timestamptz)
ON CONFLICT (id) DO UPDATE SET
email = EXCLUDED.email, full_name = EXCLUDED.full_name,
role = EXCLUDED.role, account_type = EXCLUDED.account_type,
created_date = EXCLUDED.created_date, updated_date = now()`,
id, legacy, orgID,
stringOr(u["email"], ""), stringOr(u["full_name"], ""),
stringOr(u["role"], "admin"), stringOr(u["account_type"], "employer"), created)
if err != nil {
return 0, err
}
prefs, _ := u["preferences"].(map[string]any)
if prefs == nil {
prefs = map[string]any{}
}
extra := map[string]any{}
for k, v := range prefs {
switch k {
case "owliverDefault", "compactDensity", "emailDigest":
default:
extra[k] = v
}
}
extraJSON, err := json.Marshal(extra)
if err != nil {
return 0, err
}
_, err = tx.Exec(ctx,
`INSERT INTO user_preferences (user_id, owliver_default, compact_density, email_digest, extra)
VALUES ($1::uuid, $2::boolean, $3::boolean, $4::boolean, $5::jsonb)
ON CONFLICT (user_id) DO UPDATE SET
owliver_default = EXCLUDED.owliver_default,
compact_density = EXCLUDED.compact_density,
email_digest = EXCLUDED.email_digest,
extra = EXCLUDED.extra,
updated_date = now()`,
id, boolOr(prefs["owliverDefault"], true), boolOr(prefs["compactDensity"], false),
boolOr(prefs["emailDigest"], true), extraJSON)
if err != nil {
return 0, err
}
return 1, nil
}
// pruneShiftRecords deletes this organization's shift rows that this run did
// not generate.
//
// WHY THIS EXISTS, and why it is the only place the seeder deletes anything:
//
// A shift's stable id is `shift_<worker>_<NN>`, where NN counts the shift's
// position from the OLDEST end of the rolling 56-day window
// (attendanceSeed.js:190 and shifts.go:167 — the port is faithful, the scheme
// is the problem). That number is a position, not an identity, so it means a
// different date every day the window slides. Measured against a database
// seeded one day earlier: all 114 surviving ids had moved to a different date,
// and one — `shift_marcus_41` — was orphaned, because Marcus works Mon–Fri and
// a Saturday window holds 40 of his shifts rather than 41.
//
// Upsert can rewrite the rows it still generates. It has no way to remove the
// one it no longer generates, so the collection ratchets up to the historical
// maximum and never returns to the size the generator actually produces.
//
// Deleting is safe here in a way it would not be for any other collection:
// ShiftRecord is `Ops: OpList` (api-contract.md §2 — there is no
// POST /shift-records, and U1 in §11 is exactly the question of where these
// records come from), so every row is seeder-owned and no API call can create
// one. shift_records is also a leaf table: no foreign key points at it, so
// nothing cascades. Between them, this delete cannot reach data the seeder did
// not write.
//
// Note what this does NOT do: it invents no records and changes no generated
// value. After it, the collection is exactly what BuildShifts produced for
// s.now — which is what a regenerated rolling window means.
func (s *Seeder) pruneShiftRecords(ctx context.Context, tx pgx.Tx, orgID string,
records []map[string]any) (int, error) {
// A generation that produced nothing is a bug in BuildShifts, not an
// instruction to empty the table: `id <> ALL('{}')` is true for every row.
// Refuse rather than wipe.
if len(records) == 0 {
return 0, nil
}
keep := make([]string, 0, len(records))
for _, rec := range records {
legacy, _ := rec["id"].(string)
if legacy == "" {
return 0, fmt.Errorf("generated shift record has no id")
}
keep = append(keep, DeterministicUUID("ShiftRecord:"+legacy))
}
tag, err := tx.Exec(ctx,
`DELETE FROM shift_records WHERE org_id = $1::uuid AND id <> ALL($2::uuid[])`,
orgID, keep)
if err != nil {
return 0, err
}
return int(tag.RowsAffected()), nil
}
// upsertEntity writes one collection.
func (s *Seeder) upsertEntity(ctx context.Context, tx pgx.Tx, orgID, entity string, records []map[string]any) (int, error) {
table := entityTable[entity]
res, ok := domain.ResourceByTable[table]
if !ok {
return 0, fmt.Errorf("no resource descriptor for table %s", table)
}
for _, rec := range records {
if err := s.upsertRecord(ctx, tx, orgID, entity, res, rec); err != nil {
id, _ := rec["id"].(string)
return 0, fmt.Errorf("record %s: %w", id, err)
}
}
return len(records), nil
}
func (s *Seeder) upsertRecord(ctx context.Context, tx pgx.Tx, orgID, entity string,
res *domain.Resource, rec map[string]any) error {
legacy, _ := rec["id"].(string)
if legacy == "" {
return fmt.Errorf("record has no id")
}
// user_activity's primary key is a GENERATED ALWAYS AS IDENTITY bigint, not
// a uuid, so no explicit id can be supplied for it. Its stable identity is
// legacy_id, which is what the upsert conflicts on instead.
idCol, _ := res.Column("id")
generatedID := idCol.Kind != domain.KindUUID
conflictTarget := "id"
values := map[string]any{
"legacy_id": legacy,
"org_id": orgID,
}
if generatedID {
conflictTarget = "legacy_id"
} else {
values["id"] = DeterministicUUID(entity + ":" + legacy)
}
created := stringOr(rec["created_date"], iso(s.now))
values["created_date"] = created
if _, hasUpdated := res.Column("updated_date"); hasUpdated {
// The fixture carries updated_date only on job applications, where the
// gap from created_date is what buildHires reads as time-to-hire.
// Everywhere else the column is NOT NULL and the record has never been
// modified, so it takes the creation instant.
values["updated_date"] = stringOr(rec["updated_date"], created)
}
for key, value := range rec {
switch key {
case "id", "created_date", "updated_date":
continue
}
if droppedFields[key] {
continue
}
col, ok := res.Column(key)
if !ok {
return fmt.Errorf("field %q has no column on %s", key, res.Table)
}
if referenceFields[key] && col.Kind == domain.KindUUID {
str, isStr := value.(string)
if !isStr || str == "" {
values[key] = nil
continue
}
values[key] = DeterministicUUID(referencedEntity(key) + ":" + str)
continue
}
values[key] = value
}
// Deterministic column order keeps the generated SQL stable.
names := make([]string, 0, len(values))
for k := range values {
names = append(names, k)
}
sort.Strings(names)
cols := make([]string, 0, len(names))
placeholders := make([]string, 0, len(names))
updates := make([]string, 0, len(names))
args := make([]any, 0, len(names))
for _, name := range names {
col, ok := res.Column(name)
if !ok {
return fmt.Errorf("no column %q on %s", name, res.Table)
}
bound, err := bindSeedValue(*col, values[name])
if err != nil {
return err
}
args = append(args, bound)
cols = append(cols, name)
placeholders = append(placeholders, fmt.Sprintf("$%d::%s", len(args), col.PGType))
if name != conflictTarget {
updates = append(updates, fmt.Sprintf("%s = EXCLUDED.%s", name, name))
}
}
q := fmt.Sprintf(
"INSERT INTO %s (%s) VALUES (%s) ON CONFLICT (%s) DO UPDATE SET %s",
res.Table, strings.Join(cols, ", "), strings.Join(placeholders, ", "),
conflictTarget, strings.Join(updates, ", "))
_, err := tx.Exec(ctx, q, args...)
return err
}
// referencedEntity says which entity a reference column points at, so the
// derived UUID is built from the same namespace the target was written with.
func referencedEntity(field string) string {
switch field {
case "job_posting_id", "position_id":
return "JobPosting"
case "application_id":
return "JobApplication"
case "course_id":
return "Course"
case "staff_id":
return "Staff"
case "assignment_id":
return "Assignment"
case "worker_profile_id", "candidate_id":
return "WorkerProfile"
case "interview_id":
return "AIInterview"
case "user_id", "created_by":
return "User"
}
return ""
}
// bindSeedValue converts a fixture value into something pgx can send. It is
// deliberately separate from the repository's binder: the seeder writes
// server-owned columns (id, legacy_id, org_id, created_date) that the API never
// accepts from a client.
func bindSeedValue(col domain.Column, v any) (any, error) {
if v == nil {
return nil, nil
}
switch col.Kind {
case domain.KindTextArray:
switch t := v.(type) {
case []any:
out := make([]string, 0, len(t))
for _, e := range t {
s, ok := e.(string)
if !ok {
return nil, fmt.Errorf("%s: expected an array of strings", col.Name)
}
out = append(out, s)
}
return out, nil
case []string:
return t, nil
}
return nil, fmt.Errorf("%s: expected an array", col.Name)
case domain.KindJSON:
raw, err := json.Marshal(v)
if err != nil {
return nil, fmt.Errorf("%s: %w", col.Name, err)
}
return raw, nil
case domain.KindInt:
switch t := v.(type) {
case float64:
return int64(t), nil
case int:
return int64(t), nil
case int64:
return t, nil
}
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
case domain.KindFloat:
switch t := v.(type) {
case float64:
return t, nil
case int:
return float64(t), nil
}
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
case domain.KindBool:
if b, ok := v.(bool); ok {
return b, nil
}
return nil, fmt.Errorf("%s: expected a boolean, got %T", col.Name, v)
default:
if s, ok := v.(string); ok {
return s, nil
}
return nil, fmt.Errorf("%s: expected a string, got %T", col.Name, v)
}
}
func stringOr(v any, fallback string) string {
if s, ok := v.(string); ok && s != "" {
return s
}
return fallback
}
func boolOr(v any, fallback bool) bool {
if b, ok := v.(bool); ok {
return b
}
return fallback
}

View File

@@ -0,0 +1,300 @@
package seeder_test
import (
"context"
"testing"
"time"
"github.com/krow/krow-backend/go-api/internal/seeder"
"github.com/krow/krow-backend/go-api/internal/testutil"
)
// tableFor maps the fixture's entity names onto their tables, so the counts
// asserted below come from the frontend's own data rather than from literals.
var tableFor = map[string]string{
"JobPosting": "job_postings", "JobApplication": "job_applications",
"AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles",
"Course": "courses", "Badge": "badges", "LearningPath": "learning_paths",
"Certification": "certifications", "RoleCategory": "role_categories",
"UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments",
}
func count(t *testing.T, h *testutil.Harness, table string) int {
t.Helper()
var n int
if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil {
t.Fatalf("count %s: %v", table, err)
}
return n
}
// TestSeedMatchesFixtureCounts checks every entity against the fixture rather
// than against a hardcoded headline number.
func TestSeedMatchesFixtureCounts(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
for entity, table := range tableFor {
want := len(fx.Entities[entity])
if got := count(t, h, table); got != want {
t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want)
}
}
}
// TestSeedRegressionAnchors pins the figures the demo dataset is built to
// produce. These are verified against the source, not assumed: the prompt's
// "6 postings / 22 applications" is 6 *active* postings and 24 applications.
func TestSeedRegressionAnchors(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
var active int
if err := h.Pool.QueryRow(ctx,
"SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil {
t.Fatal(err)
}
if active != 6 {
t.Errorf("active postings = %d, want 6", active)
}
if total := count(t, h, "job_postings"); total != 8 {
t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total)
}
if total := count(t, h, "job_applications"); total != 24 {
t.Errorf("applications = %d, want 24", total)
}
var scored int
var avgScored float64
if err := h.Pool.QueryRow(ctx,
"SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0").
Scan(&scored, &avgScored); err != nil {
t.Fatal(err)
}
if scored != 9 {
t.Errorf("scored applications = %d, want 9", scored)
}
if avgScored < 75.95 || avgScored > 76.05 {
t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored)
}
var hires int
var avgHire float64
if err := h.Pool.QueryRow(ctx,
"SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil {
t.Fatal(err)
}
if hires != 3 {
t.Errorf("hires = %d, want 3", hires)
}
if avgHire < 94.0 || avgHire > 94.5 {
t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire)
}
}
// TestSeedPreservesSourceValues compares stored rows field-by-field against the
// fixture, rather than trusting that the counts lining up means the data did.
func TestSeedPreservesSourceValues(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
for _, want := range fx.Entities["JobPosting"] {
legacy := want["id"].(string)
var title, status, company, roleCategory, createdDate string
var payMin, payMax int
err := h.Pool.QueryRow(ctx, `
SELECT title, status::text, company, role_category, pay_range_min, pay_range_max,
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
FROM job_postings WHERE legacy_id = $1`, legacy).
Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate)
if err != nil {
t.Fatalf("%s: %v", legacy, err)
}
if title != want["title"] {
t.Errorf("%s title = %q, want %q", legacy, title, want["title"])
}
if status != want["status"] {
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
}
if createdDate != want["created_date"] {
t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"])
}
if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) {
t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v))
}
if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) {
t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v))
}
}
// Applications carry updated_date in the source, and the gap from
// created_date is what buildHires reads as time-to-hire.
for _, want := range fx.Entities["JobApplication"] {
legacy := want["id"].(string)
var name, email, status, created, updated string
var score int
err := h.Pool.QueryRow(ctx, `
SELECT applicant_name, email::text, status::text, ai_score,
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'),
to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
FROM job_applications WHERE legacy_id = $1`, legacy).
Scan(&name, &email, &status, &score, &created, &updated)
if err != nil {
t.Fatalf("%s: %v", legacy, err)
}
if name != want["applicant_name"] {
t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"])
}
if status != want["status"] {
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
}
if created != want["created_date"] {
t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"])
}
if updated != want["updated_date"] {
t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"])
}
if v, ok := want["ai_score"].(float64); ok && score != int(v) {
t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v))
}
}
}
// TestSeedIsIdempotent runs the seeder a second time over an already-seeded
// database and expects every count and every id to be unchanged.
func TestSeedIsIdempotent(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
before := map[string]int{}
for _, table := range tableFor {
before[table] = count(t, h, table)
}
var idsBefore string
if err := h.Pool.QueryRow(ctx,
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
Scan(&idsBefore); err != nil {
t.Fatal(err)
}
if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil {
t.Fatalf("second seed: %v", err)
}
for _, table := range tableFor {
if got := count(t, h, table); got != before[table] {
t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows",
table, got, before[table])
}
}
var idsAfter string
if err := h.Pool.QueryRow(ctx,
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
Scan(&idsAfter); err != nil {
t.Fatal(err)
}
if idsAfter != idsBefore {
t.Error("application ids changed across a re-seed; keys are not deterministic")
}
}
// TestSeedRelationships checks that every reference was rewritten to a real row.
func TestSeedRelationships(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
dangling := []struct{ name, query string }{
{"applications without a posting",
`SELECT count(*) FROM job_applications a
LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`},
{"interviews without an application",
`SELECT count(*) FROM ai_interviews i
LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`},
{"staff without an application",
`SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id
WHERE s.application_id IS NOT NULL AND a.id IS NULL`},
{"shifts without staff",
`SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id
WHERE r.staff_id IS NOT NULL AND s.id IS NULL`},
{"evidence without a course",
`SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id
WHERE e.course_id IS NOT NULL AND c.id IS NULL`},
}
for _, d := range dangling {
var n int
if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil {
t.Fatalf("%s: %v", d.name, err)
}
if n != 0 {
t.Errorf("%s: %d", d.name, n)
}
}
// interview_id is a soft reference on purpose: the source contains one
// dangling value (app_devon -> int_devon), and preserving it is the point.
var set, resolve int
if err := h.Pool.QueryRow(ctx,
`SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL),
(SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`).
Scan(&set, &resolve); err != nil {
t.Fatal(err)
}
if set != 5 {
t.Errorf("applications carrying interview_id = %d, want 5", set)
}
if resolve != 4 {
t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve)
}
}
// TestSeedOrganizationScope checks every seeded row belongs to the development
// organization, so organization scoping has something real to filter on.
func TestSeedOrganizationScope(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
for _, table := range tableFor {
var wrong int
if err := h.Pool.QueryRow(ctx,
"SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID).
Scan(&wrong); err != nil {
t.Fatalf("%s: %v", table, err)
}
if wrong != 0 {
t.Errorf("%s: %d rows outside the development organization", table, wrong)
}
}
}
// TestDeterministicUUID pins the key derivation: the same source id must always
// produce the same key, or a re-seed would duplicate every row.
func TestDeterministicUUID(t *testing.T) {
a := seeder.DeterministicUUID("JobPosting:job_chef")
b := seeder.DeterministicUUID("JobPosting:job_chef")
if a != b {
t.Fatalf("not deterministic: %s != %s", a, b)
}
if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a {
t.Fatal("distinct source ids produced the same key")
}
if len(a) != 36 || a[14] != '5' {
t.Errorf("expected a v5 UUID, got %q", a)
}
}
// TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must
// produce identical shift records.
func TestSeedShiftsStableForAnchor(t *testing.T) {
anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local)
a := seeder.BuildShifts(anchor)
b := seeder.BuildShifts(anchor)
if len(a) != len(b) {
t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b))
}
for i := range a {
if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] {
t.Fatalf("shift %d differs between runs", i)
}
}
}

View File

@@ -0,0 +1,215 @@
package seeder
import (
"fmt"
"math"
"sort"
"time"
)
// A port of src/api/attendanceSeed.js.
//
// Ported rather than snapshotted because this is the one collection whose dates
// are anchored to *now* rather than to a fixed calendar. `dataResolver.inPeriod`
// windows every collection on created_date, so "attendance last week" has to
// mean last week on the day the seeder runs. A frozen JSON snapshot would read
// as permanently empty a fortnight later.
//
// Nothing here is random. The distribution is deterministic given the date the
// seeder runs, so the same day always produces the same figures and the
// regression tests can assert against them:
//
// Marco — the control: reliable, weekend event overtime
// Marcus — attendance degrading over the last fortnight (the anomaly)
// Antoine — present throughout, overtime climbing week on week (the trend)
const windowDays = 56
type rosterEntry struct {
staffID, workerName, workerEmail string
jobPostingID, role, roleCategory string
weekdays []time.Weekday
startHour int
scheduledHours float64
}
var roster = []rosterEntry{
{
staffID: "staff_marco", workerName: "Marco Rivera", workerEmail: "marco.rivera@email.com",
jobPostingID: "job_bartender_corp", role: "Experienced Bartender – Corporate Events",
roleCategory: "Bartender",
// Wed–Sat: corporate events run late in the week.
weekdays: []time.Weekday{time.Wednesday, time.Thursday, time.Friday, time.Saturday},
startHour: 16, scheduledHours: 8,
},
{
staffID: "staff_marcus", workerName: "Marcus Williams", workerEmail: "marcus.w@email.com",
jobPostingID: "job_security", role: "Event Security Officer", roleCategory: "Security",
// Mon–Fri: a fixed rota, which is what makes the recent absences stand
// out rather than read as an irregular schedule.
weekdays: []time.Weekday{time.Monday, time.Tuesday, time.Wednesday, time.Thursday, time.Friday},
startHour: 14, scheduledHours: 8,
},
{
staffID: "staff_antoine", workerName: "Chef Antoine Dubois", workerEmail: "antoine.dubois@email.com",
jobPostingID: "job_chef", role: "Executive Chef – Catering", roleCategory: "Chef",
// Tue–Sat: kitchen service.
weekdays: []time.Weekday{time.Tuesday, time.Wednesday, time.Thursday, time.Friday, time.Saturday},
startHour: 12, scheduledHours: 9,
},
}
type behaviour struct {
status string
minutesLate int
overtime float64
notes string
}
// behaviourFor mirrors the BEHAVIOUR map. `i` counts back from the most recent
// shift, so "the last fortnight" stays a range of small indices as the window
// rolls forward.
func behaviourFor(staffID string, i int, weekday time.Weekday) behaviour {
switch staffID {
case "staff_marco":
b := behaviour{status: "present"}
if i == 14 {
b.status, b.minutesLate = "late", 9
}
// Friday and Saturday events overrun; midweek ones do not.
if weekday == time.Friday || weekday == time.Saturday {
b.overtime = 1
}
return b
case "staff_marcus":
switch i {
case 2, 7:
return behaviour{status: "absent", notes: "Called in sick"}
case 4:
return behaviour{status: "no_show", notes: "No contact"}
case 1:
return behaviour{status: "late", minutesLate: 24}
case 5:
return behaviour{status: "late", minutesLate: 16}
case 9:
return behaviour{status: "late", minutesLate: 12}
case 26:
return behaviour{status: "late", minutesLate: 7}
}
return behaviour{status: "present"}
case "staff_antoine":
weekIndex := i / 5
busy := weekday == time.Thursday || weekday == time.Friday || weekday == time.Saturday
b := behaviour{status: "present"}
if busy {
b.overtime = math.Max(0.5, round1(3.5-float64(weekIndex)*0.45))
}
return b
}
return behaviour{status: "present"}
}
// round1 and round2 reproduce JavaScript's Math.round, which rounds halves away
// from zero — the same rule as Go's math.Round.
func round1(n float64) float64 { return math.Round(n*10) / 10 }
func round2(n float64) float64 { return math.Round(n*100) / 100 }
// daysAgo is `n` days back at a given local hour.
//
// Local rather than UTC because a shift belongs to the day it was worked in the
// place it was worked, and periodRange windows on local day boundaries too.
func daysAgo(now time.Time, n, hour int) time.Time {
d := now.AddDate(0, 0, -n)
return time.Date(d.Year(), d.Month(), d.Day(), hour, 0, 0, 0, now.Location())
}
func containsWeekday(set []time.Weekday, w time.Weekday) bool {
for _, x := range set {
if x == w {
return true
}
}
return false
}
// shiftOffsets is every day offset in the window on which this worker is rostered.
func shiftOffsets(now time.Time, weekdays []time.Weekday) []int {
var offsets []int
for offset := 0; offset <= windowDays; offset++ {
if containsWeekday(weekdays, daysAgo(now, offset, 0).Weekday()) {
offsets = append(offsets, offset)
}
}
return offsets
}
const isoMillis = "2006-01-02T15:04:05.000Z"
func iso(t time.Time) string { return t.UTC().Format(isoMillis) }
// BuildShifts generates the shift records for a given instant, most recent first.
func BuildShifts(now time.Time) []map[string]any {
records := make([]map[string]any, 0, 128)
for _, w := range roster {
offsets := shiftOffsets(now, w.weekdays)
for i, offset := range offsets {
scheduledStart := daysAgo(now, offset, w.startHour)
weekday := scheduledStart.Weekday()
scheduledEnd := scheduledStart.Add(time.Duration(w.scheduledHours * float64(time.Hour)))
b := behaviourFor(w.staffID, i, weekday)
worked := b.status != "absent" && b.status != "no_show"
var actualStart, actualEnd any
endForUpdated := scheduledEnd
if worked {
actualStart = iso(scheduledStart.Add(time.Duration(b.minutesLate) * time.Minute))
ae := scheduledEnd.Add(time.Duration(b.overtime * float64(time.Hour)))
actualEnd, endForUpdated = iso(ae), ae
}
actualHours, overtimeHours, minutesLate := 0.0, 0.0, 0
if worked {
actualHours = round2(w.scheduledHours - float64(b.minutesLate)/60 + b.overtime)
overtimeHours = round1(b.overtime)
minutesLate = b.minutesLate
}
records = append(records, map[string]any{
"id": fmt.Sprintf("shift_%s_%02d", w.staffID[len("staff_"):], len(offsets)-i),
"staff_id": w.staffID,
"worker_name": w.workerName,
"worker_email": w.workerEmail,
"job_posting_id": w.jobPostingID,
"role": w.role,
"role_category": w.roleCategory,
"shift_date": scheduledStart.Format("2006-01-02"),
"scheduled_start": iso(scheduledStart),
"scheduled_end": iso(scheduledEnd),
"scheduled_hours": w.scheduledHours,
"actual_start": actualStart,
"actual_end": actualEnd,
"actual_hours": actualHours,
"overtime_hours": overtimeHours,
"minutes_late": minutesLate,
"status": b.status,
"notes": b.notes,
// Load-bearing: dataResolver windows every collection on
// created_date, so a shift's created date IS the instant it
// was worked.
"created_date": iso(scheduledStart),
"updated_date": iso(endForUpdated),
})
}
}
// Most recent first, matching the -created_date order every other
// collection is listed in.
sort.SliceStable(records, func(a, b int) bool {
return records[a]["created_date"].(string) > records[b]["created_date"].(string)
})
return records
}

View File

@@ -0,0 +1,178 @@
package seeder_test
import (
"context"
"testing"
"time"
"github.com/krow/krow-backend/go-api/internal/seeder"
"github.com/krow/krow-backend/go-api/internal/testutil"
)
// The shift collection is a rolling 56-day window, and its stable id encodes a
// shift's POSITION in that window rather than its identity. Seed on Friday and
// re-seed on Saturday and every id means a different date; one of them —
// Marcus's, because he works Mon–Fri and a Saturday window holds one fewer of
// his shifts — is no longer generated at all.
//
// Upsert cannot express that. These tests pin the behaviour that can: after any
// seed, shift_records holds exactly what BuildShifts produced for that instant,
// and nothing left over from a previous run.
func shiftCount(t *testing.T, h *testutil.Harness) int {
t.Helper()
var n int
if err := h.Pool.QueryRow(context.Background(),
"SELECT count(*) FROM shift_records").Scan(&n); err != nil {
t.Fatalf("count shift_records: %v", err)
}
return n
}
// TestReseedOnALaterDayLeavesNoStaleShifts is the regression itself: the
// database was seeded on one day, the frontend regenerates on the next, and the
// two must still describe the same collection.
func TestReseedOnALaterDayLeavesNoStaleShifts(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
// A Friday, then the Saturday after it — the exact pair that orphaned
// shift_marcus_41 in the live database.
friday := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
saturday := friday.AddDate(0, 0, 1)
if _, err := seeder.New(h.Pool, fx, friday).Run(ctx); err != nil {
t.Fatalf("seed on the Friday: %v", err)
}
fridayRows := shiftCount(t, h)
if want := len(seeder.BuildShifts(friday)); fridayRows != want {
t.Fatalf("after the Friday seed: %d rows, generator produced %d", fridayRows, want)
}
result, err := seeder.New(h.Pool, fx, saturday).Run(ctx)
if err != nil {
t.Fatalf("re-seed on the Saturday: %v", err)
}
generated := seeder.BuildShifts(saturday)
if got := shiftCount(t, h); got != len(generated) {
t.Errorf("after re-seeding a day later: %d rows, but the generator produced %d "+
"— %d stale record(s) survived the re-seed", got, len(generated), got-len(generated))
}
if result.Pruned != fridayRows-len(generated) {
t.Errorf("Pruned = %d, want %d", result.Pruned, fridayRows-len(generated))
}
// Every surviving row must be one this run generated, holding this run's
// date for that id — not the previous run's.
want := map[string]string{}
for _, rec := range generated {
want[rec["id"].(string)] = rec["shift_date"].(string)
}
rows, err := h.Pool.Query(ctx, "SELECT legacy_id, shift_date::text FROM shift_records")
if err != nil {
t.Fatal(err)
}
defer rows.Close()
for rows.Next() {
var legacy, date string
if err := rows.Scan(&legacy, &date); err != nil {
t.Fatal(err)
}
switch expected, generatedNow := want[legacy]; {
case !generatedNow:
t.Errorf("%s is in the database but was not generated for this instant", legacy)
case expected != date:
t.Errorf("%s holds %s, but this run generated it as %s", legacy, date, expected)
}
}
if err := rows.Err(); err != nil {
t.Fatal(err)
}
}
// TestReseedIsConvergentAcrossAWeek walks a whole week, so the assertion does
// not depend on the one day pair that happened to expose the bug. Every day of
// the week changes the roster composition differently.
func TestReseedIsConvergentAcrossAWeek(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
day := time.Date(2026, 8, 17, 9, 0, 0, 0, time.Local) // a Monday
for i := 0; i < 7; i++ {
now := day.AddDate(0, 0, i)
if _, err := seeder.New(h.Pool, fx, now).Run(ctx); err != nil {
t.Fatalf("seed on %s: %v", now.Weekday(), err)
}
want := len(seeder.BuildShifts(now))
if got := shiftCount(t, h); got != want {
t.Errorf("%s %s: %d rows, generator produced %d",
now.Weekday(), now.Format("2006-01-02"), got, want)
}
}
}
// Re-seeding the same instant twice must still change nothing — the prune must
// not delete rows it just wrote.
func TestReseedSameInstantPrunesNothing(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
before := shiftCount(t, h)
result, err := seeder.New(h.Pool, fx, h.Now).Run(ctx)
if err != nil {
t.Fatalf("re-seed: %v", err)
}
if result.Pruned != 0 {
t.Errorf("re-seeding the same instant pruned %d record(s), want 0", result.Pruned)
}
if got := shiftCount(t, h); got != before {
t.Errorf("shift_records went from %d to %d rows on an identical re-seed", before, got)
}
}
// The prune is scoped to the organization being seeded. Another organization's
// shift records are none of its business — and once authentication lands, that
// is the difference between a re-seed and an incident.
func TestPruneIsScopedToTheSeededOrganization(t *testing.T) {
h := testutil.New(t)
fx := testutil.Fixture(t)
ctx := context.Background()
var otherOrg string
if err := h.Pool.QueryRow(ctx,
`INSERT INTO organizations (name, slug) VALUES ('Other', 'other-org') RETURNING id::text`).
Scan(&otherOrg); err != nil {
t.Fatal(err)
}
// Copy one of this organization's shifts into the other one, with an id and
// legacy_id no generation will ever produce.
if _, err := h.Pool.Exec(ctx,
`INSERT INTO shift_records (org_id, legacy_id, staff_id, worker_name, worker_email,
job_posting_id, role, role_category, shift_date, scheduled_start, scheduled_end,
scheduled_hours, actual_start, actual_end, actual_hours, overtime_hours,
minutes_late, status, notes, created_date, updated_date)
SELECT $1::uuid, 'shift_other_99', staff_id, worker_name, worker_email,
job_posting_id, role, role_category, shift_date, scheduled_start, scheduled_end,
scheduled_hours, actual_start, actual_end, actual_hours, overtime_hours,
minutes_late, status, notes, created_date, updated_date
FROM shift_records LIMIT 1`, otherOrg); err != nil {
t.Fatal(err)
}
if _, err := seeder.New(h.Pool, fx, h.Now.AddDate(0, 0, 1)).Run(ctx); err != nil {
t.Fatalf("re-seed: %v", err)
}
var survived int
if err := h.Pool.QueryRow(ctx,
"SELECT count(*) FROM shift_records WHERE org_id = $1::uuid", otherOrg).Scan(&survived); err != nil {
t.Fatal(err)
}
if survived != 1 {
t.Errorf("the other organization's shift record was pruned: %d survived, want 1", survived)
}
}

View File

@@ -0,0 +1,134 @@
package seeder_test
import (
"testing"
"time"
"github.com/krow/krow-backend/go-api/internal/seeder"
)
// The shift generator is a port of src/api/attendanceSeed.js. These tests pin
// the distribution that module's own documentation describes, so a drift in the
// port shows up as a failing assertion rather than as quietly different
// attendance figures.
func shiftsFor(anchor time.Time, email string) []map[string]any {
var out []map[string]any
for _, r := range seeder.BuildShifts(anchor) {
if r["worker_email"] == email {
out = append(out, r)
}
}
return out
}
func TestShiftDistributionMatchesSource(t *testing.T) {
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
all := seeder.BuildShifts(anchor)
counts := map[string]int{}
for _, r := range all {
counts[r["status"].(string)]++
}
// Marcus alone supplies the attendance anomaly: two absences, one no-show
// and three late arrivals in the recent window, plus one older late.
if counts["absent"] != 2 {
t.Errorf("absent = %d, want 2", counts["absent"])
}
if counts["no_show"] != 1 {
t.Errorf("no_show = %d, want 1", counts["no_show"])
}
// Marcus i=1,5,9,26 plus Marco i=14.
if counts["late"] != 5 {
t.Errorf("late = %d, want 5", counts["late"])
}
if counts["present"] == 0 {
t.Error("no present shifts generated")
}
}
func TestShiftRosterIsThreePeople(t *testing.T) {
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
emails := map[string]bool{}
for _, r := range seeder.BuildShifts(anchor) {
emails[r["worker_email"].(string)] = true
}
if len(emails) != 3 {
t.Fatalf("roster has %d people, want 3 (one per hire)", len(emails))
}
}
// A missed shift is zero hours worked, not a short one — and the schema's
// shift_records_absence_has_no_hours constraint depends on it.
func TestAbsencesHaveNoHours(t *testing.T) {
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
for _, r := range seeder.BuildShifts(anchor) {
status := r["status"].(string)
if status != "absent" && status != "no_show" {
continue
}
if r["actual_hours"].(float64) != 0 {
t.Errorf("%s: %s shift has actual_hours %v", r["id"], status, r["actual_hours"])
}
if r["minutes_late"].(int) != 0 {
t.Errorf("%s: %s shift has minutes_late %v", r["id"], status, r["minutes_late"])
}
if r["actual_start"] != nil || r["actual_end"] != nil {
t.Errorf("%s: %s shift has actual timestamps", r["id"], status)
}
}
}
// Antoine's overtime climbs week on week — a trend rather than a spike. It is
// the overtime anomaly the analytics are shaped to surface.
func TestAntoineOvertimeClimbs(t *testing.T) {
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
shifts := shiftsFor(anchor, "antoine.dubois@email.com")
if len(shifts) == 0 {
t.Fatal("no shifts generated for Antoine")
}
// BuildShifts returns most-recent-first, so recent overtime should exceed
// the overtime from the far end of the window.
var recent, older float64
for i, r := range shifts {
ot := r["overtime_hours"].(float64)
if i < 10 {
recent += ot
}
if i >= len(shifts)-10 {
older += ot
}
}
if recent <= older {
t.Errorf("overtime is not climbing: recent 10 = %.1fh, oldest 10 = %.1fh", recent, older)
}
}
// created_date is the instant the shift was worked. dataResolver windows every
// collection on it, so a shift dated anywhere else would vanish from every
// period reading.
func TestShiftCreatedDateIsTheShiftInstant(t *testing.T) {
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
for _, r := range seeder.BuildShifts(anchor) {
if r["created_date"] != r["scheduled_start"] {
t.Fatalf("%s: created_date %v is not the scheduled start %v",
r["id"], r["created_date"], r["scheduled_start"])
}
}
}
// The window rolls forward with the anchor: shifts must stay recent relative to
// whenever the seeder runs, which is the whole reason this is a port rather
// than a frozen snapshot.
func TestShiftWindowFollowsTheAnchor(t *testing.T) {
early := seeder.BuildShifts(time.Date(2026, 3, 1, 12, 0, 0, 0, time.Local))
late := seeder.BuildShifts(time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local))
if early[0]["created_date"] == late[0]["created_date"] {
t.Fatal("shift dates did not move with the anchor")
}
if got := late[0]["created_date"].(string)[:4]; got != "2026" {
t.Errorf("most recent shift is dated %s", got)
}
}