Files
krow_backend/go-api/internal/seeder/seeder.go
Suriyakumarvijayanayagam 629d97181d
Some checks failed
CI / test (push) Has been cancelled
CI / fixture (push) Has been cancelled
Rebase Staff too, which was the last thing keeping the chart incomplete
And correct the previous commit's closing claim, which was wrong.

c378bc0 said the Hiring activity chart "still renders empty ... the fault is
further down in that component, not in the data." That was a retraction of a
correct diagnosis, made from a screenshot taken before the rebase had reached
the browser. The chart was empty BECAUSE the data was stale, exactly as first
diagnosed, and rebasing fixed it. Checked properly this time: the area path
carries real values, and the rendered chart shows applications peaking at 16
around 8/18 with the screening and interview series drawn over it.

What was genuinely still missing was Hires. Staff was not rebased, so
hire_date stayed 35 days old with nothing inside the 30-day window and that
series drew nothing. It is rebased now, anchored on hire_date rather than
created_date, because the hire is the event the chart plots.

Leaving it behind had also introduced an inconsistency of my own making: once
applications moved, a candidate was hired last week according to their
application and five weeks ago according to their staff record. Rebasing them
together removes that.

All four series now render. Full suite green.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PJvibeSc1JYXjatankqM1g
2026-08-29 15:21:14 +05:30

553 lines
18 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Package seeder loads the frontend's demo dataset into PostgreSQL.
//
// Source of truth is the frontend, not this package. `seed/fixtures/seed.json`
// is produced by executing src/api/seed.js through Vite and serialising what it
// exports, so ids, dates, numbers and enum values arrive exactly as the demo
// has them — no transcription step, and nothing to drift. Shift records are the
// one exception: they are generated (see shifts.go) because their dates are
// anchored to now.
//
// Idempotency strategy: EXPLICIT UPSERT inside a single transaction.
//
// Every record's primary key is derived deterministically from its source id
// (uuid v5 over a fixed namespace), so re-running the seeder targets exactly the
// same rows and `ON CONFLICT (id) DO UPDATE` restores each one to its seeded
// values. Records created through the API survive a re-seed. A column the
// fixture does not carry is left as it is.
//
// Shift records are the one collection that is also PRUNED — see
// pruneShiftRecords. Upsert alone cannot converge a rolling window, and shift
// records are the only collection that is a rolling window.
package seeder
import (
"context"
"crypto/sha1"
"encoding/json"
"fmt"
"os"
"sort"
"strings"
"time"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgxpool"
"github.com/krow/krow-backend/go-api/internal/domain"
"github.com/krow/krow-backend/go-api/internal/orgctx"
)
// namespace is a fixed UUID used to derive record ids from source ids. Changing
// it re-keys the entire dataset, so it is a constant, not configuration.
var namespace = [16]byte{
0x6b, 0x72, 0x6f, 0x77, 0x2d, 0x73, 0x65, 0x65,
0x64, 0x2d, 0x76, 0x31, 0x00, 0x00, 0x00, 0x01,
}
// DeterministicUUID derives a stable v5 UUID from a source id.
func DeterministicUUID(name string) string {
h := sha1.New()
h.Write(namespace[:])
h.Write([]byte(name))
var b [16]byte
copy(b[:], h.Sum(nil))
b[6] = (b[6] & 0x0f) | 0x50 // version 5
b[8] = (b[8] & 0x3f) | 0x80 // RFC 4122 variant
return fmt.Sprintf("%x-%x-%x-%x-%x", b[0:4], b[4:6], b[6:8], b[8:10], b[10:16])
}
// Fixture is the serialised frontend dataset.
type Fixture struct {
DemoUser map[string]any `json:"demoUser"`
Entities map[string][]map[string]any `json:"entities"`
}
// Result counts what was written, by entity.
type Result struct {
OrgID string
Counts map[string]int
// Pruned is how many stale shift records this run removed. Reported rather
// than silent: a delete during a seed should never be something you have to
// read the source to discover.
Pruned int
}
// entityOrder is insertion order, chosen so every foreign key is satisfied by
// the time it is referenced.
var entityOrder = []string{
"RoleCategory", "Certification", "Badge", "Course", "LearningPath",
"JobPosting", "WorkerProfile", "JobApplication", "AIInterview", "Staff",
"Assignment", "ShiftRecord", "Evidence", "UserActivity",
}
// entityTable maps a frontend entity name to its table.
var entityTable = map[string]string{
"RoleCategory": "role_categories", "Certification": "certifications",
"Badge": "badges", "Course": "courses", "LearningPath": "learning_paths",
"JobPosting": "job_postings", "WorkerProfile": "worker_profiles",
"JobApplication": "job_applications", "AIInterview": "ai_interviews",
"Staff": "staff", "Assignment": "assignments", "ShiftRecord": "shift_records",
"Evidence": "evidence", "UserActivity": "user_activity",
}
// referenceFields are columns holding a source id that must be rewritten to the
// derived UUID. Every one of these is a real reference in the frontend data.
var referenceFields = map[string]bool{
"job_posting_id": true, "application_id": true, "course_id": true,
"staff_id": true, "assignment_id": true, "worker_profile_id": true,
"interview_id": true, "position_id": true, "candidate_id": true,
"user_id": true, "created_by": true,
}
// droppedFields are keys the fixture carries that no column exists for and no
// frontend code reads. Dropping them is deliberate and recorded here rather
// than being silent.
//
// _order — a positional index used only while seed.js builds its course list
// (src/api/seed.js:1326). Nothing reads it.
var droppedFields = map[string]bool{"_order": true}
// Seeder loads a fixture into a database.
type Seeder struct {
pool *pgxpool.Pool
fixture *Fixture
now time.Time
}
// Load reads a fixture from disk.
func Load(path string) (*Fixture, error) {
raw, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("read fixture %s: %w", path, err)
}
var f Fixture
if err := json.Unmarshal(raw, &f); err != nil {
return nil, fmt.Errorf("parse fixture %s: %w", path, err)
}
return &f, nil
}
// rebasedEntities names the entities whose dates are moved to sit against now,
// and the field to move them by.
//
// ShiftRecord is absent because it is GENERATED against now rather than
// rebased (see shifts.go). Everything not listed here is reference data —
// courses, role categories, badges — where a date is a fact about the record
// rather than a position in a window, and moving it would be a lie.
var rebasedEntities = map[string]string{
"UserActivity": "created_date",
"JobApplication": "created_date",
"AIInterview": "created_date",
// Anchored on hire_date, not created_date: the hire is the event the
// Hiring activity chart plots, and leaving it behind while applications
// moved produced a workspace where somebody was hired last week according
// to their application and five weeks ago according to their staff record.
"Staff": "hire_date",
}
// New builds a seeder. `now` anchors the generated shift records.
func New(pool *pgxpool.Pool, fixture *Fixture, now time.Time) *Seeder {
return &Seeder{pool: pool, fixture: fixture, now: now}
}
// Run seeds everything in one transaction: either the whole dataset lands or
// none of it does.
func (s *Seeder) Run(ctx context.Context) (*Result, error) {
tx, err := s.pool.Begin(ctx)
if err != nil {
return nil, err
}
defer func() { _ = tx.Rollback(ctx) }()
orgID, err := s.upsertOrganization(ctx, tx)
if err != nil {
return nil, err
}
result := &Result{OrgID: orgID, Counts: map[string]int{}}
n, err := s.upsertUser(ctx, tx, orgID)
if err != nil {
return nil, err
}
result.Counts["User"] = n
for _, entity := range entityOrder {
records := s.fixture.Entities[entity]
if entity == "ShiftRecord" {
records = BuildShifts(s.now)
}
// Time-series entities are authored on a fixed calendar and would
// otherwise age out of every window the product asks about — the
// activity tools, and the hiring chart on Control Center. See
// RebaseToNow. Each is anchored on its OWN newest record, so the gaps
// within an entity are preserved without inventing a relationship
// between entities.
if field, ok := rebasedEntities[entity]; ok {
records = RebaseToNow(records, field, s.now)
}
count, err := s.upsertEntity(ctx, tx, orgID, entity, records)
if err != nil {
return nil, fmt.Errorf("seed %s: %w", entity, err)
}
result.Counts[entity] = count
if entity == "ShiftRecord" {
pruned, err := s.pruneShiftRecords(ctx, tx, orgID, records)
if err != nil {
return nil, fmt.Errorf("prune ShiftRecord: %w", err)
}
result.Pruned = pruned
}
}
if err := tx.Commit(ctx); err != nil {
return nil, err
}
return result, nil
}
// upsertOrganization creates the development organization the whole dataset
// belongs to. See internal/orgctx — this is not a tenant, it is a placeholder
// with a stable id so re-seeding is idempotent.
func (s *Seeder) upsertOrganization(ctx context.Context, tx pgx.Tx) (string, error) {
id := DeterministicUUID("org:" + orgctx.DevOrgSlug)
_, err := tx.Exec(ctx,
`INSERT INTO organizations (id, name, slug) VALUES ($1::uuid, $2, $3::citext)
ON CONFLICT (id) DO UPDATE SET name = EXCLUDED.name, updated_date = now()`,
id, orgctx.DevOrgName, orgctx.DevOrgSlug)
return id, err
}
// upsertUser writes the demo user and splits its preferences into their own
// table, as api-contract.md §9 describes.
func (s *Seeder) upsertUser(ctx context.Context, tx pgx.Tx, orgID string) (int, error) {
u := s.fixture.DemoUser
if u == nil {
return 0, nil
}
legacy, _ := u["id"].(string)
id := DeterministicUUID("User:" + legacy)
created := stringOr(u["created_date"], iso(s.now))
_, err := tx.Exec(ctx,
`INSERT INTO users (id, legacy_id, org_id, email, full_name, role, account_type, created_date, updated_date)
VALUES ($1::uuid, $2::text, $3::uuid, $4::citext, $5::text, $6::text, $7::text, $8::timestamptz, $8::timestamptz)
ON CONFLICT (id) DO UPDATE SET
email = EXCLUDED.email, full_name = EXCLUDED.full_name,
role = EXCLUDED.role, account_type = EXCLUDED.account_type,
created_date = EXCLUDED.created_date, updated_date = now()`,
id, legacy, orgID,
stringOr(u["email"], ""), stringOr(u["full_name"], ""),
stringOr(u["role"], "admin"), stringOr(u["account_type"], "employer"), created)
if err != nil {
return 0, err
}
prefs, _ := u["preferences"].(map[string]any)
if prefs == nil {
prefs = map[string]any{}
}
extra := map[string]any{}
for k, v := range prefs {
switch k {
case "owliverDefault", "compactDensity", "emailDigest":
default:
extra[k] = v
}
}
extraJSON, err := json.Marshal(extra)
if err != nil {
return 0, err
}
_, err = tx.Exec(ctx,
`INSERT INTO user_preferences (user_id, owliver_default, compact_density, email_digest, extra)
VALUES ($1::uuid, $2::boolean, $3::boolean, $4::boolean, $5::jsonb)
ON CONFLICT (user_id) DO UPDATE SET
owliver_default = EXCLUDED.owliver_default,
compact_density = EXCLUDED.compact_density,
email_digest = EXCLUDED.email_digest,
extra = EXCLUDED.extra,
updated_date = now()`,
id, boolOr(prefs["owliverDefault"], true), boolOr(prefs["compactDensity"], false),
boolOr(prefs["emailDigest"], true), extraJSON)
if err != nil {
return 0, err
}
return 1, nil
}
// pruneShiftRecords deletes this organization's shift rows that this run did
// not generate.
//
// WHY THIS EXISTS, and why it is the only place the seeder deletes anything:
//
// A shift's stable id is `shift_<worker>_<NN>`, where NN counts the shift's
// position from the OLDEST end of the rolling 56-day window
// (attendanceSeed.js:190 and shifts.go:167 — the port is faithful, the scheme
// is the problem). That number is a position, not an identity, so it means a
// different date every day the window slides. Measured against a database
// seeded one day earlier: all 114 surviving ids had moved to a different date,
// and one — `shift_marcus_41` — was orphaned, because Marcus works Mon–Fri and
// a Saturday window holds 40 of his shifts rather than 41.
//
// Upsert can rewrite the rows it still generates. It has no way to remove the
// one it no longer generates, so the collection ratchets up to the historical
// maximum and never returns to the size the generator actually produces.
//
// Deleting is safe here in a way it would not be for any other collection:
// ShiftRecord is `Ops: OpList` (api-contract.md §2 — there is no
// POST /shift-records, and U1 in §11 is exactly the question of where these
// records come from), so every row is seeder-owned and no API call can create
// one. shift_records is also a leaf table: no foreign key points at it, so
// nothing cascades. Between them, this delete cannot reach data the seeder did
// not write.
//
// Note what this does NOT do: it invents no records and changes no generated
// value. After it, the collection is exactly what BuildShifts produced for
// s.now — which is what a regenerated rolling window means.
func (s *Seeder) pruneShiftRecords(ctx context.Context, tx pgx.Tx, orgID string,
records []map[string]any) (int, error) {
// A generation that produced nothing is a bug in BuildShifts, not an
// instruction to empty the table: `id <> ALL('{}')` is true for every row.
// Refuse rather than wipe.
if len(records) == 0 {
return 0, nil
}
keep := make([]string, 0, len(records))
for _, rec := range records {
legacy, _ := rec["id"].(string)
if legacy == "" {
return 0, fmt.Errorf("generated shift record has no id")
}
keep = append(keep, DeterministicUUID("ShiftRecord:"+legacy))
}
tag, err := tx.Exec(ctx,
`DELETE FROM shift_records WHERE org_id = $1::uuid AND id <> ALL($2::uuid[])`,
orgID, keep)
if err != nil {
return 0, err
}
return int(tag.RowsAffected()), nil
}
// upsertEntity writes one collection.
func (s *Seeder) upsertEntity(ctx context.Context, tx pgx.Tx, orgID, entity string, records []map[string]any) (int, error) {
table := entityTable[entity]
res, ok := domain.ResourceByTable[table]
if !ok {
return 0, fmt.Errorf("no resource descriptor for table %s", table)
}
for _, rec := range records {
if err := s.upsertRecord(ctx, tx, orgID, entity, res, rec); err != nil {
id, _ := rec["id"].(string)
return 0, fmt.Errorf("record %s: %w", id, err)
}
}
return len(records), nil
}
func (s *Seeder) upsertRecord(ctx context.Context, tx pgx.Tx, orgID, entity string,
res *domain.Resource, rec map[string]any) error {
legacy, _ := rec["id"].(string)
if legacy == "" {
return fmt.Errorf("record has no id")
}
// user_activity's primary key is a GENERATED ALWAYS AS IDENTITY bigint, not
// a uuid, so no explicit id can be supplied for it. Its stable identity is
// legacy_id, which is what the upsert conflicts on instead.
idCol, _ := res.Column("id")
generatedID := idCol.Kind != domain.KindUUID
conflictTarget := "id"
values := map[string]any{
"legacy_id": legacy,
"org_id": orgID,
}
if generatedID {
conflictTarget = "legacy_id"
} else {
values["id"] = DeterministicUUID(entity + ":" + legacy)
}
created := stringOr(rec["created_date"], iso(s.now))
values["created_date"] = created
if _, hasUpdated := res.Column("updated_date"); hasUpdated {
// The fixture carries updated_date only on job applications, where the
// gap from created_date is what buildHires reads as time-to-hire.
// Everywhere else the column is NOT NULL and the record has never been
// modified, so it takes the creation instant.
values["updated_date"] = stringOr(rec["updated_date"], created)
}
for key, value := range rec {
switch key {
case "id", "created_date", "updated_date":
continue
}
if droppedFields[key] {
continue
}
col, ok := res.Column(key)
if !ok {
return fmt.Errorf("field %q has no column on %s", key, res.Table)
}
if referenceFields[key] && col.Kind == domain.KindUUID {
str, isStr := value.(string)
if !isStr || str == "" {
values[key] = nil
continue
}
values[key] = DeterministicUUID(referencedEntity(key) + ":" + str)
continue
}
values[key] = value
}
// Deterministic column order keeps the generated SQL stable.
names := make([]string, 0, len(values))
for k := range values {
names = append(names, k)
}
sort.Strings(names)
cols := make([]string, 0, len(names))
placeholders := make([]string, 0, len(names))
updates := make([]string, 0, len(names))
args := make([]any, 0, len(names))
for _, name := range names {
col, ok := res.Column(name)
if !ok {
return fmt.Errorf("no column %q on %s", name, res.Table)
}
bound, err := bindSeedValue(*col, values[name])
if err != nil {
return err
}
args = append(args, bound)
cols = append(cols, name)
placeholders = append(placeholders, fmt.Sprintf("$%d::%s", len(args), col.PGType))
if name != conflictTarget {
updates = append(updates, fmt.Sprintf("%s = EXCLUDED.%s", name, name))
}
}
q := fmt.Sprintf(
"INSERT INTO %s (%s) VALUES (%s) ON CONFLICT (%s) DO UPDATE SET %s",
res.Table, strings.Join(cols, ", "), strings.Join(placeholders, ", "),
conflictTarget, strings.Join(updates, ", "))
_, err := tx.Exec(ctx, q, args...)
return err
}
// referencedEntity says which entity a reference column points at, so the
// derived UUID is built from the same namespace the target was written with.
func referencedEntity(field string) string {
switch field {
case "job_posting_id", "position_id":
return "JobPosting"
case "application_id":
return "JobApplication"
case "course_id":
return "Course"
case "staff_id":
return "Staff"
case "assignment_id":
return "Assignment"
case "worker_profile_id", "candidate_id":
return "WorkerProfile"
case "interview_id":
return "AIInterview"
case "user_id", "created_by":
return "User"
}
return ""
}
// bindSeedValue converts a fixture value into something pgx can send. It is
// deliberately separate from the repository's binder: the seeder writes
// server-owned columns (id, legacy_id, org_id, created_date) that the API never
// accepts from a client.
func bindSeedValue(col domain.Column, v any) (any, error) {
if v == nil {
return nil, nil
}
switch col.Kind {
case domain.KindTextArray:
switch t := v.(type) {
case []any:
out := make([]string, 0, len(t))
for _, e := range t {
s, ok := e.(string)
if !ok {
return nil, fmt.Errorf("%s: expected an array of strings", col.Name)
}
out = append(out, s)
}
return out, nil
case []string:
return t, nil
}
return nil, fmt.Errorf("%s: expected an array", col.Name)
case domain.KindJSON:
raw, err := json.Marshal(v)
if err != nil {
return nil, fmt.Errorf("%s: %w", col.Name, err)
}
return raw, nil
case domain.KindInt:
switch t := v.(type) {
case float64:
return int64(t), nil
case int:
return int64(t), nil
case int64:
return t, nil
}
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
case domain.KindFloat:
switch t := v.(type) {
case float64:
return t, nil
case int:
return float64(t), nil
}
return nil, fmt.Errorf("%s: expected a number, got %T", col.Name, v)
case domain.KindBool:
if b, ok := v.(bool); ok {
return b, nil
}
return nil, fmt.Errorf("%s: expected a boolean, got %T", col.Name, v)
default:
if s, ok := v.(string); ok {
return s, nil
}
return nil, fmt.Errorf("%s: expected a string, got %T", col.Name, v)
}
}
func stringOr(v any, fallback string) string {
if s, ok := v.(string); ok && s != "" {
return s
}
return fallback
}
func boolOr(v any, fallback bool) bool {
if b, ok := v.(bool); ok {
return b
}
return fallback
}