first commit
This commit is contained in:
525
go-api/internal/seeder/seeder.go
Normal file
525
go-api/internal/seeder/seeder.go
Normal file
@@ -0,0 +1,525 @@
|
||||
// Package seeder loads the frontend's demo dataset into PostgreSQL.
|
||||
//
|
||||
// Source of truth is the frontend, not this package. `seed/fixtures/seed.json`
|
||||
// is produced by executing src/api/seed.js through Vite and serialising what it
|
||||
// exports, so ids, dates, numbers and enum values arrive exactly as the demo
|
||||
// has them — no transcription step, and nothing to drift. Shift records are the
|
||||
// one exception: they are generated (see shifts.go) because their dates are
|
||||
// anchored to now.
|
||||
//
|
||||
// Idempotency strategy: EXPLICIT UPSERT inside a single transaction.
|
||||
//
|
||||
// Every record's primary key is derived deterministically from its source id
|
||||
// (uuid v5 over a fixed namespace), so re-running the seeder targets exactly the
|
||||
// same rows and `ON CONFLICT (id) DO UPDATE` restores each one to its seeded
|
||||
// values. Records created through the API survive a re-seed. A column the
|
||||
// fixture does not carry is left as it is.
|
||||
//
|
||||
// Shift records are the one collection that is also PRUNED — see
|
||||
// pruneShiftRecords. Upsert alone cannot converge a rolling window, and shift
|
||||
// records are the only collection that is a rolling window.
|
||||
package seeder
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/sha1"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5"
|
||||
"github.com/jackc/pgx/v5/pgxpool"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/domain"
|
||||
"github.com/krow/krow-backend/go-api/internal/orgctx"
|
||||
)
|
||||
|
||||
// namespace is a fixed UUID used to derive record ids from source ids. Changing
|
||||
// it re-keys the entire dataset, so it is a constant, not configuration.
|
||||
var namespace = [16]byte{
|
||||
0x6b, 0x72, 0x6f, 0x77, 0x2d, 0x73, 0x65, 0x65,
|
||||
0x64, 0x2d, 0x76, 0x31, 0x00, 0x00, 0x00, 0x01,
|
||||
}
|
||||
|
||||
// DeterministicUUID derives a stable v5 UUID from a source id.
|
||||
func DeterministicUUID(name string) string {
|
||||
h := sha1.New()
|
||||
h.Write(namespace[:])
|
||||
h.Write([]byte(name))
|
||||
var b [16]byte
|
||||
copy(b[:], h.Sum(nil))
|
||||
b[6] = (b[6] & 0x0f) | 0x50 // version 5
|
||||
b[8] = (b[8] & 0x3f) | 0x80 // RFC 4122 variant
|
||||
return fmt.Sprintf("%x-%x-%x-%x-%x", b[0:4], b[4:6], b[6:8], b[8:10], b[10:16])
|
||||
}
|
||||
|
||||
// Fixture is the serialised frontend dataset.
|
||||
type Fixture struct {
|
||||
DemoUser map[string]any `json:"demoUser"`
|
||||
Entities map[string][]map[string]any `json:"entities"`
|
||||
}
|
||||
|
||||
// Result counts what was written, by entity.
|
||||
type Result struct {
|
||||
OrgID string
|
||||
Counts map[string]int
|
||||
// Pruned is how many stale shift records this run removed. Reported rather
|
||||
// than silent: a delete during a seed should never be something you have to
|
||||
// read the source to discover.
|
||||
Pruned int
|
||||
}
|
||||
|
||||
// entityOrder is insertion order, chosen so every foreign key is satisfied by
|
||||
// the time it is referenced.
|
||||
var entityOrder = []string{
|
||||
"RoleCategory", "Certification", "Badge", "Course", "LearningPath",
|
||||
"JobPosting", "WorkerProfile", "JobApplication", "AIInterview", "Staff",
|
||||
"Assignment", "ShiftRecord", "Evidence", "UserActivity",
|
||||
}
|
||||
|
||||
// entityTable maps a frontend entity name to its table.
|
||||
var entityTable = map[string]string{
|
||||
"RoleCategory": "role_categories", "Certification": "certifications",
|
||||
"Badge": "badges", "Course": "courses", "LearningPath": "learning_paths",
|
||||
"JobPosting": "job_postings", "WorkerProfile": "worker_profiles",
|
||||
"JobApplication": "job_applications", "AIInterview": "ai_interviews",
|
||||
"Staff": "staff", "Assignment": "assignments", "ShiftRecord": "shift_records",
|
||||
"Evidence": "evidence", "UserActivity": "user_activity",
|
||||
}
|
||||
|
||||
// referenceFields are columns holding a source id that must be rewritten to the
|
||||
// derived UUID. Every one of these is a real reference in the frontend data.
|
||||
var referenceFields = map[string]bool{
|
||||
"job_posting_id": true, "application_id": true, "course_id": true,
|
||||
"staff_id": true, "assignment_id": true, "worker_profile_id": true,
|
||||
"interview_id": true, "position_id": true, "candidate_id": true,
|
||||
"user_id": true, "created_by": true,
|
||||
}
|
||||
|
||||
// droppedFields are keys the fixture carries that no column exists for and no
|
||||
// frontend code reads. Dropping them is deliberate and recorded here rather
|
||||
// than being silent.
|
||||
//
|
||||
// _order — a positional index used only while seed.js builds its course list
|
||||
// (src/api/seed.js:1326). Nothing reads it.
|
||||
var droppedFields = map[string]bool{"_order": true}
|
||||
|
||||
// Seeder loads a fixture into a database.
|
||||
type Seeder struct {
|
||||
pool *pgxpool.Pool
|
||||
fixture *Fixture
|
||||
now time.Time
|
||||
}
|
||||
|
||||
// Load reads a fixture from disk.
|
||||
func Load(path string) (*Fixture, error) {
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read fixture %s: %w", path, err)
|
||||
}
|
||||
var f Fixture
|
||||
if err := json.Unmarshal(raw, &f); err != nil {
|
||||
return nil, fmt.Errorf("parse fixture %s: %w", path, err)
|
||||
}
|
||||
return &f, nil
|
||||
}
|
||||
|
||||
// New builds a seeder. `now` anchors the generated shift records.
|
||||
func New(pool *pgxpool.Pool, fixture *Fixture, now time.Time) *Seeder {
|
||||
return &Seeder{pool: pool, fixture: fixture, now: now}
|
||||
}
|
||||
|
||||
// Run seeds everything in one transaction: either the whole dataset lands or
|
||||
// none of it does.
|
||||
func (s *Seeder) Run(ctx context.Context) (*Result, error) {
|
||||
tx, err := s.pool.Begin(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer func() { _ = tx.Rollback(ctx) }()
|
||||
|
||||
orgID, err := s.upsertOrganization(ctx, tx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result := &Result{OrgID: orgID, Counts: map[string]int{}}
|
||||
|
||||
n, err := s.upsertUser(ctx, tx, orgID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
result.Counts["User"] = n
|
||||
|
||||
for _, entity := range entityOrder {
|
||||
records := s.fixture.Entities[entity]
|
||||
if entity == "ShiftRecord" {
|
||||
records = BuildShifts(s.now)
|
||||
}
|
||||
count, err := s.upsertEntity(ctx, tx, orgID, entity, records)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("seed %s: %w", entity, err)
|
||||
}
|
||||
result.Counts[entity] = count
|
||||
|
||||
if entity == "ShiftRecord" {
|
||||
pruned, err := s.pruneShiftRecords(ctx, tx, orgID, records)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("prune ShiftRecord: %w", err)
|
||||
}
|
||||
result.Pruned = pruned
|
||||
}
|
||||
}
|
||||
|
||||
if err := tx.Commit(ctx); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// upsertOrganization creates the development organization the whole dataset
|
||||
// belongs to. See internal/orgctx — this is not a tenant, it is a placeholder
|
||||
// with a stable id so re-seeding is idempotent.
|
||||
func (s *Seeder) upsertOrganization(ctx context.Context, tx pgx.Tx) (string, error) {
|
||||
id := DeterministicUUID("org:" + orgctx.DevOrgSlug)
|
||||
_, err := tx.Exec(ctx,
|
||||
`INSERT INTO organizations (id, name, slug) VALUES ($1::uuid, $2, $3::citext)
|
||||
ON CONFLICT (id) DO UPDATE SET name = EXCLUDED.name, updated_date = now()`,
|
||||
id, orgctx.DevOrgName, orgctx.DevOrgSlug)
|
||||
return id, err
|
||||
}
|
||||
|
||||
// upsertUser writes the demo user and splits its preferences into their own
|
||||
// table, as api-contract.md §9 describes.
|
||||
func (s *Seeder) upsertUser(ctx context.Context, tx pgx.Tx, orgID string) (int, error) {
|
||||
u := s.fixture.DemoUser
|
||||
if u == nil {
|
||||
return 0, nil
|
||||
}
|
||||
legacy, _ := u["id"].(string)
|
||||
id := DeterministicUUID("User:" + legacy)
|
||||
created := stringOr(u["created_date"], iso(s.now))
|
||||
|
||||
_, err := tx.Exec(ctx,
|
||||
`INSERT INTO users (id, legacy_id, org_id, email, full_name, role, account_type, created_date, updated_date)
|
||||
VALUES ($1::uuid, $2::text, $3::uuid, $4::citext, $5::text, $6::text, $7::text, $8::timestamptz, $8::timestamptz)
|
||||
ON CONFLICT (id) DO UPDATE SET
|
||||
email = EXCLUDED.email, full_name = EXCLUDED.full_name,
|
||||
role = EXCLUDED.role, account_type = EXCLUDED.account_type,
|
||||
created_date = EXCLUDED.created_date, updated_date = now()`,
|
||||
id, legacy, orgID,
|
||||
stringOr(u["email"], ""), stringOr(u["full_name"], ""),
|
||||
stringOr(u["role"], "admin"), stringOr(u["account_type"], "employer"), created)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
prefs, _ := u["preferences"].(map[string]any)
|
||||
if prefs == nil {
|
||||
prefs = map[string]any{}
|
||||
}
|
||||
extra := map[string]any{}
|
||||
for k, v := range prefs {
|
||||
switch k {
|
||||
case "owliverDefault", "compactDensity", "emailDigest":
|
||||
default:
|
||||
extra[k] = v
|
||||
}
|
||||
}
|
||||
extraJSON, err := json.Marshal(extra)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
_, err = tx.Exec(ctx,
|
||||
`INSERT INTO user_preferences (user_id, owliver_default, compact_density, email_digest, extra)
|
||||
VALUES ($1::uuid, $2::boolean, $3::boolean, $4::boolean, $5::jsonb)
|
||||
ON CONFLICT (user_id) DO UPDATE SET
|
||||
owliver_default = EXCLUDED.owliver_default,
|
||||
compact_density = EXCLUDED.compact_density,
|
||||
email_digest = EXCLUDED.email_digest,
|
||||
extra = EXCLUDED.extra,
|
||||
updated_date = now()`,
|
||||
id, boolOr(prefs["owliverDefault"], true), boolOr(prefs["compactDensity"], false),
|
||||
boolOr(prefs["emailDigest"], true), extraJSON)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return 1, nil
|
||||
}
|
||||
|
||||
// pruneShiftRecords deletes this organization's shift rows that this run did
|
||||
// not generate.
|
||||
//
|
||||
// WHY THIS EXISTS, and why it is the only place the seeder deletes anything:
|
||||
//
|
||||
// A shift's stable id is `shift_<worker>_<NN>`, where NN counts the shift's
|
||||
// position from the OLDEST end of the rolling 56-day window
|
||||
// (attendanceSeed.js:190 and shifts.go:167 — the port is faithful, the scheme
|
||||
// is the problem). That number is a position, not an identity, so it means a
|
||||
// different date every day the window slides. Measured against a database
|
||||
// seeded one day earlier: all 114 surviving ids had moved to a different date,
|
||||
// and one — `shift_marcus_41` — was orphaned, because Marcus works Mon–Fri and
|
||||
// a Saturday window holds 40 of his shifts rather than 41.
|
||||
//
|
||||
// Upsert can rewrite the rows it still generates. It has no way to remove the
|
||||
// one it no longer generates, so the collection ratchets up to the historical
|
||||
// maximum and never returns to the size the generator actually produces.
|
||||
//
|
||||
// Deleting is safe here in a way it would not be for any other collection:
|
||||
// ShiftRecord is `Ops: OpList` (api-contract.md §2 — there is no
|
||||
// POST /shift-records, and U1 in §11 is exactly the question of where these
|
||||
// records come from), so every row is seeder-owned and no API call can create
|
||||
// one. shift_records is also a leaf table: no foreign key points at it, so
|
||||
// nothing cascades. Between them, this delete cannot reach data the seeder did
|
||||
// not write.
|
||||
//
|
||||
// Note what this does NOT do: it invents no records and changes no generated
|
||||
// value. After it, the collection is exactly what BuildShifts produced for
|
||||
// s.now — which is what a regenerated rolling window means.
|
||||
func (s *Seeder) pruneShiftRecords(ctx context.Context, tx pgx.Tx, orgID string,
|
||||
records []map[string]any) (int, error) {
|
||||
|
||||
// A generation that produced nothing is a bug in BuildShifts, not an
|
||||
// instruction to empty the table: `id <> ALL('{}')` is true for every row.
|
||||
// Refuse rather than wipe.
|
||||
if len(records) == 0 {
|
||||
return 0, nil
|
||||
}
|
||||
|
||||
keep := make([]string, 0, len(records))
|
||||
for _, rec := range records {
|
||||
legacy, _ := rec["id"].(string)
|
||||
if legacy == "" {
|
||||
return 0, fmt.Errorf("generated shift record has no id")
|
||||
}
|
||||
keep = append(keep, DeterministicUUID("ShiftRecord:"+legacy))
|
||||
}
|
||||
|
||||
tag, err := tx.Exec(ctx,
|
||||
`DELETE FROM shift_records WHERE org_id = $1::uuid AND id <> ALL($2::uuid[])`,
|
||||
orgID, keep)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return int(tag.RowsAffected()), nil
|
||||
}
|
||||
|
||||
// upsertEntity writes one collection.
|
||||
func (s *Seeder) upsertEntity(ctx context.Context, tx pgx.Tx, orgID, entity string, records []map[string]any) (int, error) {
|
||||
table := entityTable[entity]
|
||||
res, ok := domain.ResourceByTable[table]
|
||||
if !ok {
|
||||
return 0, fmt.Errorf("no resource descriptor for table %s", table)
|
||||
}
|
||||
for _, rec := range records {
|
||||
if err := s.upsertRecord(ctx, tx, orgID, entity, res, rec); err != nil {
|
||||
id, _ := rec["id"].(string)
|
||||
return 0, fmt.Errorf("record %s: %w", id, err)
|
||||
}
|
||||
}
|
||||
return len(records), nil
|
||||
}
|
||||
|
||||
func (s *Seeder) upsertRecord(ctx context.Context, tx pgx.Tx, orgID, entity string,
|
||||
res *domain.Resource, rec map[string]any) error {
|
||||
|
||||
legacy, _ := rec["id"].(string)
|
||||
if legacy == "" {
|
||||
return fmt.Errorf("record has no id")
|
||||
}
|
||||
|
||||
// user_activity's primary key is a GENERATED ALWAYS AS IDENTITY bigint, not
|
||||
// a uuid, so no explicit id can be supplied for it. Its stable identity is
|
||||
// legacy_id, which is what the upsert conflicts on instead.
|
||||
idCol, _ := res.Column("id")
|
||||
generatedID := idCol.Kind != domain.KindUUID
|
||||
conflictTarget := "id"
|
||||
|
||||
values := map[string]any{
|
||||
"legacy_id": legacy,
|
||||
"org_id": orgID,
|
||||
}
|
||||
if generatedID {
|
||||
conflictTarget = "legacy_id"
|
||||
} else {
|
||||
values["id"] = DeterministicUUID(entity + ":" + legacy)
|
||||
}
|
||||
|
||||
created := stringOr(rec["created_date"], iso(s.now))
|
||||
values["created_date"] = created
|
||||
if _, hasUpdated := res.Column("updated_date"); hasUpdated {
|
||||
// The fixture carries updated_date only on job applications, where the
|
||||
// gap from created_date is what buildHires reads as time-to-hire.
|
||||
// Everywhere else the column is NOT NULL and the record has never been
|
||||
// modified, so it takes the creation instant.
|
||||
values["updated_date"] = stringOr(rec["updated_date"], created)
|
||||
}
|
||||
|
||||
for key, value := range rec {
|
||||
switch key {
|
||||
case "id", "created_date", "updated_date":
|
||||
continue
|
||||
}
|
||||
if droppedFields[key] {
|
||||
continue
|
||||
}
|
||||
col, ok := res.Column(key)
|
||||
if !ok {
|
||||
return fmt.Errorf("field %q has no column on %s", key, res.Table)
|
||||
}
|
||||
if referenceFields[key] && col.Kind == domain.KindUUID {
|
||||
str, isStr := value.(string)
|
||||
if !isStr || str == "" {
|
||||
values[key] = nil
|
||||
continue
|
||||
}
|
||||
values[key] = DeterministicUUID(referencedEntity(key) + ":" + str)
|
||||
continue
|
||||
}
|
||||
values[key] = value
|
||||
}
|
||||
|
||||
// Deterministic column order keeps the generated SQL stable.
|
||||
names := make([]string, 0, len(values))
|
||||
for k := range values {
|
||||
names = append(names, k)
|
||||
}
|
||||
sort.Strings(names)
|
||||
|
||||
cols := make([]string, 0, len(names))
|
||||
placeholders := make([]string, 0, len(names))
|
||||
updates := make([]string, 0, len(names))
|
||||
args := make([]any, 0, len(names))
|
||||
|
||||
for _, name := range names {
|
||||
col, ok := res.Column(name)
|
||||
if !ok {
|
||||
return fmt.Errorf("no column %q on %s", name, res.Table)
|
||||
}
|
||||
bound, err := bindSeedValue(*col, values[name])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
args = append(args, bound)
|
||||
cols = append(cols, name)
|
||||
placeholders = append(placeholders, fmt.Sprintf("$%d::%s", len(args), col.PGType))
|
||||
if name != conflictTarget {
|
||||
updates = append(updates, fmt.Sprintf("%s = EXCLUDED.%s", name, name))
|
||||
}
|
||||
}
|
||||
|
||||
q := fmt.Sprintf(
|
||||
"INSERT INTO %s (%s) VALUES (%s) ON CONFLICT (%s) DO UPDATE SET %s",
|
||||
res.Table, strings.Join(cols, ", "), strings.Join(placeholders, ", "),
|
||||
conflictTarget, strings.Join(updates, ", "))
|
||||
|
||||
_, err := tx.Exec(ctx, q, args...)
|
||||
return err
|
||||
}
|
||||
|
||||
// referencedEntity says which entity a reference column points at, so the
|
||||
// derived UUID is built from the same namespace the target was written with.
|
||||
func referencedEntity(field string) string {
|
||||
switch field {
|
||||
case "job_posting_id", "position_id":
|
||||
return "JobPosting"
|
||||
case "application_id":
|
||||
return "JobApplication"
|
||||
case "course_id":
|
||||
return "Course"
|
||||
case "staff_id":
|
||||
return "Staff"
|
||||
case "assignment_id":
|
||||
return "Assignment"
|
||||
case "worker_profile_id", "candidate_id":
|
||||
return "WorkerProfile"
|
||||
case "interview_id":
|
||||
return "AIInterview"
|
||||
case "user_id", "created_by":
|
||||
return "User"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// bindSeedValue converts a fixture value into something pgx can send. It is
|
||||
// deliberately separate from the repository's binder: the seeder writes
|
||||
// server-owned columns (id, legacy_id, org_id, created_date) that the API never
|
||||
// accepts from a client.
|
||||
func bindSeedValue(col domain.Column, v any) (any, error) {
|
||||
if v == nil {
|
||||
return nil, nil
|
||||
}
|
||||
switch col.Kind {
|
||||
case domain.KindTextArray:
|
||||
switch t := v.(type) {
|
||||
case []any:
|
||||
out := make([]string, 0, len(t))
|
||||
for _, e := range t {
|
||||
s, ok := e.(string)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: expected an array of strings", col.Name)
|
||||
}
|
||||
out = append(out, s)
|
||||
}
|
||||
return out, nil
|
||||
case []string:
|
||||
return t, nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: expected an array", col.Name)
|
||||
|
||||
case domain.KindJSON:
|
||||
raw, err := json.Marshal(v)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", col.Name, err)
|
||||
}
|
||||
return raw, nil
|
||||
|
||||
case domain.KindInt:
|
||||
switch t := v.(type) {
|
||||
case float64:
|
||||
return int64(t), nil
|
||||
case int:
|
||||
return int64(t), nil
|
||||
case int64:
|
||||
return t, nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
|
||||
|
||||
case domain.KindFloat:
|
||||
switch t := v.(type) {
|
||||
case float64:
|
||||
return t, nil
|
||||
case int:
|
||||
return float64(t), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
|
||||
|
||||
case domain.KindBool:
|
||||
if b, ok := v.(bool); ok {
|
||||
return b, nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: expected a boolean, got %T", col.Name, v)
|
||||
|
||||
default:
|
||||
if s, ok := v.(string); ok {
|
||||
return s, nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: expected a string, got %T", col.Name, v)
|
||||
}
|
||||
}
|
||||
|
||||
func stringOr(v any, fallback string) string {
|
||||
if s, ok := v.(string); ok && s != "" {
|
||||
return s
|
||||
}
|
||||
return fallback
|
||||
}
|
||||
|
||||
func boolOr(v any, fallback bool) bool {
|
||||
if b, ok := v.(bool); ok {
|
||||
return b
|
||||
}
|
||||
return fallback
|
||||
}
|
||||
300
go-api/internal/seeder/seeder_test.go
Normal file
300
go-api/internal/seeder/seeder_test.go
Normal file
@@ -0,0 +1,300 @@
|
||||
package seeder_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/seeder"
|
||||
"github.com/krow/krow-backend/go-api/internal/testutil"
|
||||
)
|
||||
|
||||
// tableFor maps the fixture's entity names onto their tables, so the counts
|
||||
// asserted below come from the frontend's own data rather than from literals.
|
||||
var tableFor = map[string]string{
|
||||
"JobPosting": "job_postings", "JobApplication": "job_applications",
|
||||
"AIInterview": "ai_interviews", "Staff": "staff", "WorkerProfile": "worker_profiles",
|
||||
"Course": "courses", "Badge": "badges", "LearningPath": "learning_paths",
|
||||
"Certification": "certifications", "RoleCategory": "role_categories",
|
||||
"UserActivity": "user_activity", "Evidence": "evidence", "Assignment": "assignments",
|
||||
}
|
||||
|
||||
func count(t *testing.T, h *testutil.Harness, table string) int {
|
||||
t.Helper()
|
||||
var n int
|
||||
if err := h.Pool.QueryRow(context.Background(), "SELECT count(*) FROM "+table).Scan(&n); err != nil {
|
||||
t.Fatalf("count %s: %v", table, err)
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// TestSeedMatchesFixtureCounts checks every entity against the fixture rather
|
||||
// than against a hardcoded headline number.
|
||||
func TestSeedMatchesFixtureCounts(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
|
||||
for entity, table := range tableFor {
|
||||
want := len(fx.Entities[entity])
|
||||
if got := count(t, h, table); got != want {
|
||||
t.Errorf("%s: seeded %d rows, fixture has %d", table, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedRegressionAnchors pins the figures the demo dataset is built to
|
||||
// produce. These are verified against the source, not assumed: the prompt's
|
||||
// "6 postings / 22 applications" is 6 *active* postings and 24 applications.
|
||||
func TestSeedRegressionAnchors(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
ctx := context.Background()
|
||||
|
||||
var active int
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT count(*) FROM job_postings WHERE status = 'active'").Scan(&active); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if active != 6 {
|
||||
t.Errorf("active postings = %d, want 6", active)
|
||||
}
|
||||
if total := count(t, h, "job_postings"); total != 8 {
|
||||
t.Errorf("job postings = %d, want 8 (6 active, 1 paused, 1 closed)", total)
|
||||
}
|
||||
if total := count(t, h, "job_applications"); total != 24 {
|
||||
t.Errorf("applications = %d, want 24", total)
|
||||
}
|
||||
|
||||
var scored int
|
||||
var avgScored float64
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT count(*), coalesce(avg(ai_score), 0) FROM job_applications WHERE ai_score > 0").
|
||||
Scan(&scored, &avgScored); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if scored != 9 {
|
||||
t.Errorf("scored applications = %d, want 9", scored)
|
||||
}
|
||||
if avgScored < 75.95 || avgScored > 76.05 {
|
||||
t.Errorf("average scored ai_score = %.2f, want 76.0", avgScored)
|
||||
}
|
||||
|
||||
var hires int
|
||||
var avgHire float64
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT count(*), coalesce(avg(ai_score), 0) FROM staff").Scan(&hires, &avgHire); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if hires != 3 {
|
||||
t.Errorf("hires = %d, want 3", hires)
|
||||
}
|
||||
if avgHire < 94.0 || avgHire > 94.5 {
|
||||
t.Errorf("average hire ai_score = %.2f, want ~94.3", avgHire)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedPreservesSourceValues compares stored rows field-by-field against the
|
||||
// fixture, rather than trusting that the counts lining up means the data did.
|
||||
func TestSeedPreservesSourceValues(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
for _, want := range fx.Entities["JobPosting"] {
|
||||
legacy := want["id"].(string)
|
||||
var title, status, company, roleCategory, createdDate string
|
||||
var payMin, payMax int
|
||||
err := h.Pool.QueryRow(ctx, `
|
||||
SELECT title, status::text, company, role_category, pay_range_min, pay_range_max,
|
||||
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
||||
FROM job_postings WHERE legacy_id = $1`, legacy).
|
||||
Scan(&title, &status, &company, &roleCategory, &payMin, &payMax, &createdDate)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v", legacy, err)
|
||||
}
|
||||
if title != want["title"] {
|
||||
t.Errorf("%s title = %q, want %q", legacy, title, want["title"])
|
||||
}
|
||||
if status != want["status"] {
|
||||
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
||||
}
|
||||
if createdDate != want["created_date"] {
|
||||
t.Errorf("%s created_date = %q, want %q", legacy, createdDate, want["created_date"])
|
||||
}
|
||||
if v, ok := want["pay_range_min"].(float64); ok && payMin != int(v) {
|
||||
t.Errorf("%s pay_range_min = %d, want %d", legacy, payMin, int(v))
|
||||
}
|
||||
if v, ok := want["pay_range_max"].(float64); ok && payMax != int(v) {
|
||||
t.Errorf("%s pay_range_max = %d, want %d", legacy, payMax, int(v))
|
||||
}
|
||||
}
|
||||
|
||||
// Applications carry updated_date in the source, and the gap from
|
||||
// created_date is what buildHires reads as time-to-hire.
|
||||
for _, want := range fx.Entities["JobApplication"] {
|
||||
legacy := want["id"].(string)
|
||||
var name, email, status, created, updated string
|
||||
var score int
|
||||
err := h.Pool.QueryRow(ctx, `
|
||||
SELECT applicant_name, email::text, status::text, ai_score,
|
||||
to_char(created_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"'),
|
||||
to_char(updated_date AT TIME ZONE 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.MS"Z"')
|
||||
FROM job_applications WHERE legacy_id = $1`, legacy).
|
||||
Scan(&name, &email, &status, &score, &created, &updated)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v", legacy, err)
|
||||
}
|
||||
if name != want["applicant_name"] {
|
||||
t.Errorf("%s applicant_name = %q, want %q", legacy, name, want["applicant_name"])
|
||||
}
|
||||
if status != want["status"] {
|
||||
t.Errorf("%s status = %q, want %q", legacy, status, want["status"])
|
||||
}
|
||||
if created != want["created_date"] {
|
||||
t.Errorf("%s created_date = %q, want %q", legacy, created, want["created_date"])
|
||||
}
|
||||
if updated != want["updated_date"] {
|
||||
t.Errorf("%s updated_date = %q, want %q", legacy, updated, want["updated_date"])
|
||||
}
|
||||
if v, ok := want["ai_score"].(float64); ok && score != int(v) {
|
||||
t.Errorf("%s ai_score = %d, want %d", legacy, score, int(v))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedIsIdempotent runs the seeder a second time over an already-seeded
|
||||
// database and expects every count and every id to be unchanged.
|
||||
func TestSeedIsIdempotent(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
before := map[string]int{}
|
||||
for _, table := range tableFor {
|
||||
before[table] = count(t, h, table)
|
||||
}
|
||||
var idsBefore string
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
||||
Scan(&idsBefore); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if _, err := seeder.New(h.Pool, fx, h.Now).Run(ctx); err != nil {
|
||||
t.Fatalf("second seed: %v", err)
|
||||
}
|
||||
|
||||
for _, table := range tableFor {
|
||||
if got := count(t, h, table); got != before[table] {
|
||||
t.Errorf("%s: %d rows after re-seed, %d before — the seeder duplicated rows",
|
||||
table, got, before[table])
|
||||
}
|
||||
}
|
||||
var idsAfter string
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT coalesce(string_agg(id::text, ',' ORDER BY id), '') FROM job_applications").
|
||||
Scan(&idsAfter); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if idsAfter != idsBefore {
|
||||
t.Error("application ids changed across a re-seed; keys are not deterministic")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedRelationships checks that every reference was rewritten to a real row.
|
||||
func TestSeedRelationships(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
ctx := context.Background()
|
||||
|
||||
dangling := []struct{ name, query string }{
|
||||
{"applications without a posting",
|
||||
`SELECT count(*) FROM job_applications a
|
||||
LEFT JOIN job_postings p ON p.id = a.job_posting_id WHERE p.id IS NULL`},
|
||||
{"interviews without an application",
|
||||
`SELECT count(*) FROM ai_interviews i
|
||||
LEFT JOIN job_applications a ON a.id = i.application_id WHERE a.id IS NULL`},
|
||||
{"staff without an application",
|
||||
`SELECT count(*) FROM staff s LEFT JOIN job_applications a ON a.id = s.application_id
|
||||
WHERE s.application_id IS NOT NULL AND a.id IS NULL`},
|
||||
{"shifts without staff",
|
||||
`SELECT count(*) FROM shift_records r LEFT JOIN staff s ON s.id = r.staff_id
|
||||
WHERE r.staff_id IS NOT NULL AND s.id IS NULL`},
|
||||
{"evidence without a course",
|
||||
`SELECT count(*) FROM evidence e LEFT JOIN courses c ON c.id = e.course_id
|
||||
WHERE e.course_id IS NOT NULL AND c.id IS NULL`},
|
||||
}
|
||||
for _, d := range dangling {
|
||||
var n int
|
||||
if err := h.Pool.QueryRow(ctx, d.query).Scan(&n); err != nil {
|
||||
t.Fatalf("%s: %v", d.name, err)
|
||||
}
|
||||
if n != 0 {
|
||||
t.Errorf("%s: %d", d.name, n)
|
||||
}
|
||||
}
|
||||
|
||||
// interview_id is a soft reference on purpose: the source contains one
|
||||
// dangling value (app_devon -> int_devon), and preserving it is the point.
|
||||
var set, resolve int
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
`SELECT (SELECT count(*) FROM job_applications WHERE interview_id IS NOT NULL),
|
||||
(SELECT count(*) FROM job_applications a JOIN ai_interviews i ON i.id = a.interview_id)`).
|
||||
Scan(&set, &resolve); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if set != 5 {
|
||||
t.Errorf("applications carrying interview_id = %d, want 5", set)
|
||||
}
|
||||
if resolve != 4 {
|
||||
t.Errorf("resolvable interview_id = %d, want 4 (int_devon dangles in the source)", resolve)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedOrganizationScope checks every seeded row belongs to the development
|
||||
// organization, so organization scoping has something real to filter on.
|
||||
func TestSeedOrganizationScope(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
ctx := context.Background()
|
||||
for _, table := range tableFor {
|
||||
var wrong int
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT count(*) FROM "+table+" WHERE org_id IS DISTINCT FROM $1::uuid", h.OrgID).
|
||||
Scan(&wrong); err != nil {
|
||||
t.Fatalf("%s: %v", table, err)
|
||||
}
|
||||
if wrong != 0 {
|
||||
t.Errorf("%s: %d rows outside the development organization", table, wrong)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeterministicUUID pins the key derivation: the same source id must always
|
||||
// produce the same key, or a re-seed would duplicate every row.
|
||||
func TestDeterministicUUID(t *testing.T) {
|
||||
a := seeder.DeterministicUUID("JobPosting:job_chef")
|
||||
b := seeder.DeterministicUUID("JobPosting:job_chef")
|
||||
if a != b {
|
||||
t.Fatalf("not deterministic: %s != %s", a, b)
|
||||
}
|
||||
if c := seeder.DeterministicUUID("JobPosting:job_security"); c == a {
|
||||
t.Fatal("distinct source ids produced the same key")
|
||||
}
|
||||
if len(a) != 36 || a[14] != '5' {
|
||||
t.Errorf("expected a v5 UUID, got %q", a)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSeedDoesNotDependOnWallClock: seeding twice with the same anchor must
|
||||
// produce identical shift records.
|
||||
func TestSeedShiftsStableForAnchor(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 15, 0, 0, 0, time.Local)
|
||||
a := seeder.BuildShifts(anchor)
|
||||
b := seeder.BuildShifts(anchor)
|
||||
if len(a) != len(b) {
|
||||
t.Fatalf("shift count differs between runs: %d vs %d", len(a), len(b))
|
||||
}
|
||||
for i := range a {
|
||||
if a[i]["id"] != b[i]["id"] || a[i]["created_date"] != b[i]["created_date"] {
|
||||
t.Fatalf("shift %d differs between runs", i)
|
||||
}
|
||||
}
|
||||
}
|
||||
215
go-api/internal/seeder/shifts.go
Normal file
215
go-api/internal/seeder/shifts.go
Normal file
@@ -0,0 +1,215 @@
|
||||
package seeder
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"sort"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A port of src/api/attendanceSeed.js.
|
||||
//
|
||||
// Ported rather than snapshotted because this is the one collection whose dates
|
||||
// are anchored to *now* rather than to a fixed calendar. `dataResolver.inPeriod`
|
||||
// windows every collection on created_date, so "attendance last week" has to
|
||||
// mean last week on the day the seeder runs. A frozen JSON snapshot would read
|
||||
// as permanently empty a fortnight later.
|
||||
//
|
||||
// Nothing here is random. The distribution is deterministic given the date the
|
||||
// seeder runs, so the same day always produces the same figures and the
|
||||
// regression tests can assert against them:
|
||||
//
|
||||
// Marco — the control: reliable, weekend event overtime
|
||||
// Marcus — attendance degrading over the last fortnight (the anomaly)
|
||||
// Antoine — present throughout, overtime climbing week on week (the trend)
|
||||
|
||||
const windowDays = 56
|
||||
|
||||
type rosterEntry struct {
|
||||
staffID, workerName, workerEmail string
|
||||
jobPostingID, role, roleCategory string
|
||||
weekdays []time.Weekday
|
||||
startHour int
|
||||
scheduledHours float64
|
||||
}
|
||||
|
||||
var roster = []rosterEntry{
|
||||
{
|
||||
staffID: "staff_marco", workerName: "Marco Rivera", workerEmail: "marco.rivera@email.com",
|
||||
jobPostingID: "job_bartender_corp", role: "Experienced Bartender – Corporate Events",
|
||||
roleCategory: "Bartender",
|
||||
// Wed–Sat: corporate events run late in the week.
|
||||
weekdays: []time.Weekday{time.Wednesday, time.Thursday, time.Friday, time.Saturday},
|
||||
startHour: 16, scheduledHours: 8,
|
||||
},
|
||||
{
|
||||
staffID: "staff_marcus", workerName: "Marcus Williams", workerEmail: "marcus.w@email.com",
|
||||
jobPostingID: "job_security", role: "Event Security Officer", roleCategory: "Security",
|
||||
// Mon–Fri: a fixed rota, which is what makes the recent absences stand
|
||||
// out rather than read as an irregular schedule.
|
||||
weekdays: []time.Weekday{time.Monday, time.Tuesday, time.Wednesday, time.Thursday, time.Friday},
|
||||
startHour: 14, scheduledHours: 8,
|
||||
},
|
||||
{
|
||||
staffID: "staff_antoine", workerName: "Chef Antoine Dubois", workerEmail: "antoine.dubois@email.com",
|
||||
jobPostingID: "job_chef", role: "Executive Chef – Catering", roleCategory: "Chef",
|
||||
// Tue–Sat: kitchen service.
|
||||
weekdays: []time.Weekday{time.Tuesday, time.Wednesday, time.Thursday, time.Friday, time.Saturday},
|
||||
startHour: 12, scheduledHours: 9,
|
||||
},
|
||||
}
|
||||
|
||||
type behaviour struct {
|
||||
status string
|
||||
minutesLate int
|
||||
overtime float64
|
||||
notes string
|
||||
}
|
||||
|
||||
// behaviourFor mirrors the BEHAVIOUR map. `i` counts back from the most recent
|
||||
// shift, so "the last fortnight" stays a range of small indices as the window
|
||||
// rolls forward.
|
||||
func behaviourFor(staffID string, i int, weekday time.Weekday) behaviour {
|
||||
switch staffID {
|
||||
case "staff_marco":
|
||||
b := behaviour{status: "present"}
|
||||
if i == 14 {
|
||||
b.status, b.minutesLate = "late", 9
|
||||
}
|
||||
// Friday and Saturday events overrun; midweek ones do not.
|
||||
if weekday == time.Friday || weekday == time.Saturday {
|
||||
b.overtime = 1
|
||||
}
|
||||
return b
|
||||
|
||||
case "staff_marcus":
|
||||
switch i {
|
||||
case 2, 7:
|
||||
return behaviour{status: "absent", notes: "Called in sick"}
|
||||
case 4:
|
||||
return behaviour{status: "no_show", notes: "No contact"}
|
||||
case 1:
|
||||
return behaviour{status: "late", minutesLate: 24}
|
||||
case 5:
|
||||
return behaviour{status: "late", minutesLate: 16}
|
||||
case 9:
|
||||
return behaviour{status: "late", minutesLate: 12}
|
||||
case 26:
|
||||
return behaviour{status: "late", minutesLate: 7}
|
||||
}
|
||||
return behaviour{status: "present"}
|
||||
|
||||
case "staff_antoine":
|
||||
weekIndex := i / 5
|
||||
busy := weekday == time.Thursday || weekday == time.Friday || weekday == time.Saturday
|
||||
b := behaviour{status: "present"}
|
||||
if busy {
|
||||
b.overtime = math.Max(0.5, round1(3.5-float64(weekIndex)*0.45))
|
||||
}
|
||||
return b
|
||||
}
|
||||
return behaviour{status: "present"}
|
||||
}
|
||||
|
||||
// round1 and round2 reproduce JavaScript's Math.round, which rounds halves away
|
||||
// from zero — the same rule as Go's math.Round.
|
||||
func round1(n float64) float64 { return math.Round(n*10) / 10 }
|
||||
func round2(n float64) float64 { return math.Round(n*100) / 100 }
|
||||
|
||||
// daysAgo is `n` days back at a given local hour.
|
||||
//
|
||||
// Local rather than UTC because a shift belongs to the day it was worked in the
|
||||
// place it was worked, and periodRange windows on local day boundaries too.
|
||||
func daysAgo(now time.Time, n, hour int) time.Time {
|
||||
d := now.AddDate(0, 0, -n)
|
||||
return time.Date(d.Year(), d.Month(), d.Day(), hour, 0, 0, 0, now.Location())
|
||||
}
|
||||
|
||||
func containsWeekday(set []time.Weekday, w time.Weekday) bool {
|
||||
for _, x := range set {
|
||||
if x == w {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// shiftOffsets is every day offset in the window on which this worker is rostered.
|
||||
func shiftOffsets(now time.Time, weekdays []time.Weekday) []int {
|
||||
var offsets []int
|
||||
for offset := 0; offset <= windowDays; offset++ {
|
||||
if containsWeekday(weekdays, daysAgo(now, offset, 0).Weekday()) {
|
||||
offsets = append(offsets, offset)
|
||||
}
|
||||
}
|
||||
return offsets
|
||||
}
|
||||
|
||||
const isoMillis = "2006-01-02T15:04:05.000Z"
|
||||
|
||||
func iso(t time.Time) string { return t.UTC().Format(isoMillis) }
|
||||
|
||||
// BuildShifts generates the shift records for a given instant, most recent first.
|
||||
func BuildShifts(now time.Time) []map[string]any {
|
||||
records := make([]map[string]any, 0, 128)
|
||||
|
||||
for _, w := range roster {
|
||||
offsets := shiftOffsets(now, w.weekdays)
|
||||
for i, offset := range offsets {
|
||||
scheduledStart := daysAgo(now, offset, w.startHour)
|
||||
weekday := scheduledStart.Weekday()
|
||||
scheduledEnd := scheduledStart.Add(time.Duration(w.scheduledHours * float64(time.Hour)))
|
||||
|
||||
b := behaviourFor(w.staffID, i, weekday)
|
||||
worked := b.status != "absent" && b.status != "no_show"
|
||||
|
||||
var actualStart, actualEnd any
|
||||
endForUpdated := scheduledEnd
|
||||
if worked {
|
||||
actualStart = iso(scheduledStart.Add(time.Duration(b.minutesLate) * time.Minute))
|
||||
ae := scheduledEnd.Add(time.Duration(b.overtime * float64(time.Hour)))
|
||||
actualEnd, endForUpdated = iso(ae), ae
|
||||
}
|
||||
|
||||
actualHours, overtimeHours, minutesLate := 0.0, 0.0, 0
|
||||
if worked {
|
||||
actualHours = round2(w.scheduledHours - float64(b.minutesLate)/60 + b.overtime)
|
||||
overtimeHours = round1(b.overtime)
|
||||
minutesLate = b.minutesLate
|
||||
}
|
||||
|
||||
records = append(records, map[string]any{
|
||||
"id": fmt.Sprintf("shift_%s_%02d", w.staffID[len("staff_"):], len(offsets)-i),
|
||||
"staff_id": w.staffID,
|
||||
"worker_name": w.workerName,
|
||||
"worker_email": w.workerEmail,
|
||||
"job_posting_id": w.jobPostingID,
|
||||
"role": w.role,
|
||||
"role_category": w.roleCategory,
|
||||
"shift_date": scheduledStart.Format("2006-01-02"),
|
||||
"scheduled_start": iso(scheduledStart),
|
||||
"scheduled_end": iso(scheduledEnd),
|
||||
"scheduled_hours": w.scheduledHours,
|
||||
"actual_start": actualStart,
|
||||
"actual_end": actualEnd,
|
||||
"actual_hours": actualHours,
|
||||
"overtime_hours": overtimeHours,
|
||||
"minutes_late": minutesLate,
|
||||
"status": b.status,
|
||||
"notes": b.notes,
|
||||
// Load-bearing: dataResolver windows every collection on
|
||||
// created_date, so a shift's created date IS the instant it
|
||||
// was worked.
|
||||
"created_date": iso(scheduledStart),
|
||||
"updated_date": iso(endForUpdated),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Most recent first, matching the -created_date order every other
|
||||
// collection is listed in.
|
||||
sort.SliceStable(records, func(a, b int) bool {
|
||||
return records[a]["created_date"].(string) > records[b]["created_date"].(string)
|
||||
})
|
||||
return records
|
||||
}
|
||||
178
go-api/internal/seeder/shifts_convergence_test.go
Normal file
178
go-api/internal/seeder/shifts_convergence_test.go
Normal file
@@ -0,0 +1,178 @@
|
||||
package seeder_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/seeder"
|
||||
"github.com/krow/krow-backend/go-api/internal/testutil"
|
||||
)
|
||||
|
||||
// The shift collection is a rolling 56-day window, and its stable id encodes a
|
||||
// shift's POSITION in that window rather than its identity. Seed on Friday and
|
||||
// re-seed on Saturday and every id means a different date; one of them —
|
||||
// Marcus's, because he works Mon–Fri and a Saturday window holds one fewer of
|
||||
// his shifts — is no longer generated at all.
|
||||
//
|
||||
// Upsert cannot express that. These tests pin the behaviour that can: after any
|
||||
// seed, shift_records holds exactly what BuildShifts produced for that instant,
|
||||
// and nothing left over from a previous run.
|
||||
|
||||
func shiftCount(t *testing.T, h *testutil.Harness) int {
|
||||
t.Helper()
|
||||
var n int
|
||||
if err := h.Pool.QueryRow(context.Background(),
|
||||
"SELECT count(*) FROM shift_records").Scan(&n); err != nil {
|
||||
t.Fatalf("count shift_records: %v", err)
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// TestReseedOnALaterDayLeavesNoStaleShifts is the regression itself: the
|
||||
// database was seeded on one day, the frontend regenerates on the next, and the
|
||||
// two must still describe the same collection.
|
||||
func TestReseedOnALaterDayLeavesNoStaleShifts(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
// A Friday, then the Saturday after it — the exact pair that orphaned
|
||||
// shift_marcus_41 in the live database.
|
||||
friday := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
saturday := friday.AddDate(0, 0, 1)
|
||||
|
||||
if _, err := seeder.New(h.Pool, fx, friday).Run(ctx); err != nil {
|
||||
t.Fatalf("seed on the Friday: %v", err)
|
||||
}
|
||||
fridayRows := shiftCount(t, h)
|
||||
if want := len(seeder.BuildShifts(friday)); fridayRows != want {
|
||||
t.Fatalf("after the Friday seed: %d rows, generator produced %d", fridayRows, want)
|
||||
}
|
||||
|
||||
result, err := seeder.New(h.Pool, fx, saturday).Run(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("re-seed on the Saturday: %v", err)
|
||||
}
|
||||
|
||||
generated := seeder.BuildShifts(saturday)
|
||||
if got := shiftCount(t, h); got != len(generated) {
|
||||
t.Errorf("after re-seeding a day later: %d rows, but the generator produced %d "+
|
||||
"— %d stale record(s) survived the re-seed", got, len(generated), got-len(generated))
|
||||
}
|
||||
if result.Pruned != fridayRows-len(generated) {
|
||||
t.Errorf("Pruned = %d, want %d", result.Pruned, fridayRows-len(generated))
|
||||
}
|
||||
|
||||
// Every surviving row must be one this run generated, holding this run's
|
||||
// date for that id — not the previous run's.
|
||||
want := map[string]string{}
|
||||
for _, rec := range generated {
|
||||
want[rec["id"].(string)] = rec["shift_date"].(string)
|
||||
}
|
||||
rows, err := h.Pool.Query(ctx, "SELECT legacy_id, shift_date::text FROM shift_records")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
for rows.Next() {
|
||||
var legacy, date string
|
||||
if err := rows.Scan(&legacy, &date); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
switch expected, generatedNow := want[legacy]; {
|
||||
case !generatedNow:
|
||||
t.Errorf("%s is in the database but was not generated for this instant", legacy)
|
||||
case expected != date:
|
||||
t.Errorf("%s holds %s, but this run generated it as %s", legacy, date, expected)
|
||||
}
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReseedIsConvergentAcrossAWeek walks a whole week, so the assertion does
|
||||
// not depend on the one day pair that happened to expose the bug. Every day of
|
||||
// the week changes the roster composition differently.
|
||||
func TestReseedIsConvergentAcrossAWeek(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
day := time.Date(2026, 8, 17, 9, 0, 0, 0, time.Local) // a Monday
|
||||
for i := 0; i < 7; i++ {
|
||||
now := day.AddDate(0, 0, i)
|
||||
if _, err := seeder.New(h.Pool, fx, now).Run(ctx); err != nil {
|
||||
t.Fatalf("seed on %s: %v", now.Weekday(), err)
|
||||
}
|
||||
want := len(seeder.BuildShifts(now))
|
||||
if got := shiftCount(t, h); got != want {
|
||||
t.Errorf("%s %s: %d rows, generator produced %d",
|
||||
now.Weekday(), now.Format("2006-01-02"), got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Re-seeding the same instant twice must still change nothing — the prune must
|
||||
// not delete rows it just wrote.
|
||||
func TestReseedSameInstantPrunesNothing(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
before := shiftCount(t, h)
|
||||
result, err := seeder.New(h.Pool, fx, h.Now).Run(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("re-seed: %v", err)
|
||||
}
|
||||
if result.Pruned != 0 {
|
||||
t.Errorf("re-seeding the same instant pruned %d record(s), want 0", result.Pruned)
|
||||
}
|
||||
if got := shiftCount(t, h); got != before {
|
||||
t.Errorf("shift_records went from %d to %d rows on an identical re-seed", before, got)
|
||||
}
|
||||
}
|
||||
|
||||
// The prune is scoped to the organization being seeded. Another organization's
|
||||
// shift records are none of its business — and once authentication lands, that
|
||||
// is the difference between a re-seed and an incident.
|
||||
func TestPruneIsScopedToTheSeededOrganization(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
fx := testutil.Fixture(t)
|
||||
ctx := context.Background()
|
||||
|
||||
var otherOrg string
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
`INSERT INTO organizations (name, slug) VALUES ('Other', 'other-org') RETURNING id::text`).
|
||||
Scan(&otherOrg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Copy one of this organization's shifts into the other one, with an id and
|
||||
// legacy_id no generation will ever produce.
|
||||
if _, err := h.Pool.Exec(ctx,
|
||||
`INSERT INTO shift_records (org_id, legacy_id, staff_id, worker_name, worker_email,
|
||||
job_posting_id, role, role_category, shift_date, scheduled_start, scheduled_end,
|
||||
scheduled_hours, actual_start, actual_end, actual_hours, overtime_hours,
|
||||
minutes_late, status, notes, created_date, updated_date)
|
||||
SELECT $1::uuid, 'shift_other_99', staff_id, worker_name, worker_email,
|
||||
job_posting_id, role, role_category, shift_date, scheduled_start, scheduled_end,
|
||||
scheduled_hours, actual_start, actual_end, actual_hours, overtime_hours,
|
||||
minutes_late, status, notes, created_date, updated_date
|
||||
FROM shift_records LIMIT 1`, otherOrg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if _, err := seeder.New(h.Pool, fx, h.Now.AddDate(0, 0, 1)).Run(ctx); err != nil {
|
||||
t.Fatalf("re-seed: %v", err)
|
||||
}
|
||||
|
||||
var survived int
|
||||
if err := h.Pool.QueryRow(ctx,
|
||||
"SELECT count(*) FROM shift_records WHERE org_id = $1::uuid", otherOrg).Scan(&survived); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if survived != 1 {
|
||||
t.Errorf("the other organization's shift record was pruned: %d survived, want 1", survived)
|
||||
}
|
||||
}
|
||||
134
go-api/internal/seeder/shifts_test.go
Normal file
134
go-api/internal/seeder/shifts_test.go
Normal file
@@ -0,0 +1,134 @@
|
||||
package seeder_test
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/seeder"
|
||||
)
|
||||
|
||||
// The shift generator is a port of src/api/attendanceSeed.js. These tests pin
|
||||
// the distribution that module's own documentation describes, so a drift in the
|
||||
// port shows up as a failing assertion rather than as quietly different
|
||||
// attendance figures.
|
||||
|
||||
func shiftsFor(anchor time.Time, email string) []map[string]any {
|
||||
var out []map[string]any
|
||||
for _, r := range seeder.BuildShifts(anchor) {
|
||||
if r["worker_email"] == email {
|
||||
out = append(out, r)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestShiftDistributionMatchesSource(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
all := seeder.BuildShifts(anchor)
|
||||
|
||||
counts := map[string]int{}
|
||||
for _, r := range all {
|
||||
counts[r["status"].(string)]++
|
||||
}
|
||||
|
||||
// Marcus alone supplies the attendance anomaly: two absences, one no-show
|
||||
// and three late arrivals in the recent window, plus one older late.
|
||||
if counts["absent"] != 2 {
|
||||
t.Errorf("absent = %d, want 2", counts["absent"])
|
||||
}
|
||||
if counts["no_show"] != 1 {
|
||||
t.Errorf("no_show = %d, want 1", counts["no_show"])
|
||||
}
|
||||
// Marcus i=1,5,9,26 plus Marco i=14.
|
||||
if counts["late"] != 5 {
|
||||
t.Errorf("late = %d, want 5", counts["late"])
|
||||
}
|
||||
if counts["present"] == 0 {
|
||||
t.Error("no present shifts generated")
|
||||
}
|
||||
}
|
||||
|
||||
func TestShiftRosterIsThreePeople(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
emails := map[string]bool{}
|
||||
for _, r := range seeder.BuildShifts(anchor) {
|
||||
emails[r["worker_email"].(string)] = true
|
||||
}
|
||||
if len(emails) != 3 {
|
||||
t.Fatalf("roster has %d people, want 3 (one per hire)", len(emails))
|
||||
}
|
||||
}
|
||||
|
||||
// A missed shift is zero hours worked, not a short one — and the schema's
|
||||
// shift_records_absence_has_no_hours constraint depends on it.
|
||||
func TestAbsencesHaveNoHours(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
for _, r := range seeder.BuildShifts(anchor) {
|
||||
status := r["status"].(string)
|
||||
if status != "absent" && status != "no_show" {
|
||||
continue
|
||||
}
|
||||
if r["actual_hours"].(float64) != 0 {
|
||||
t.Errorf("%s: %s shift has actual_hours %v", r["id"], status, r["actual_hours"])
|
||||
}
|
||||
if r["minutes_late"].(int) != 0 {
|
||||
t.Errorf("%s: %s shift has minutes_late %v", r["id"], status, r["minutes_late"])
|
||||
}
|
||||
if r["actual_start"] != nil || r["actual_end"] != nil {
|
||||
t.Errorf("%s: %s shift has actual timestamps", r["id"], status)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Antoine's overtime climbs week on week — a trend rather than a spike. It is
|
||||
// the overtime anomaly the analytics are shaped to surface.
|
||||
func TestAntoineOvertimeClimbs(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
shifts := shiftsFor(anchor, "antoine.dubois@email.com")
|
||||
if len(shifts) == 0 {
|
||||
t.Fatal("no shifts generated for Antoine")
|
||||
}
|
||||
|
||||
// BuildShifts returns most-recent-first, so recent overtime should exceed
|
||||
// the overtime from the far end of the window.
|
||||
var recent, older float64
|
||||
for i, r := range shifts {
|
||||
ot := r["overtime_hours"].(float64)
|
||||
if i < 10 {
|
||||
recent += ot
|
||||
}
|
||||
if i >= len(shifts)-10 {
|
||||
older += ot
|
||||
}
|
||||
}
|
||||
if recent <= older {
|
||||
t.Errorf("overtime is not climbing: recent 10 = %.1fh, oldest 10 = %.1fh", recent, older)
|
||||
}
|
||||
}
|
||||
|
||||
// created_date is the instant the shift was worked. dataResolver windows every
|
||||
// collection on it, so a shift dated anywhere else would vanish from every
|
||||
// period reading.
|
||||
func TestShiftCreatedDateIsTheShiftInstant(t *testing.T) {
|
||||
anchor := time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local)
|
||||
for _, r := range seeder.BuildShifts(anchor) {
|
||||
if r["created_date"] != r["scheduled_start"] {
|
||||
t.Fatalf("%s: created_date %v is not the scheduled start %v",
|
||||
r["id"], r["created_date"], r["scheduled_start"])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The window rolls forward with the anchor: shifts must stay recent relative to
|
||||
// whenever the seeder runs, which is the whole reason this is a port rather
|
||||
// than a frozen snapshot.
|
||||
func TestShiftWindowFollowsTheAnchor(t *testing.T) {
|
||||
early := seeder.BuildShifts(time.Date(2026, 3, 1, 12, 0, 0, 0, time.Local))
|
||||
late := seeder.BuildShifts(time.Date(2026, 8, 21, 12, 0, 0, 0, time.Local))
|
||||
if early[0]["created_date"] == late[0]["created_date"] {
|
||||
t.Fatal("shift dates did not move with the anchor")
|
||||
}
|
||||
if got := late[0]["created_date"].(string)[:4]; got != "2026" {
|
||||
t.Errorf("most recent shift is dated %s", got)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user