diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..28a6d56 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,45 @@ +# Keep the build context small and keep secrets out of it. +# +# Everything the image needs is go-api/ and seed/fixtures/seed.json. Anything +# listed here is either a secret, a build output, or bytes that would be sent +# to the daemon and then ignored — each of which slows every build. + +# ── Secrets. Never in an image, never in a layer. ─────────────────────────── +.env +.env.* +!.env.example +*.pem +*.key +*.crt + +# ── Version control and CI ────────────────────────────────────────────────── +.git +.gitignore +.github + +# ── Build output ──────────────────────────────────────────────────────────── +go-api/bin/ +*.test +*.out +coverage.* + +# ── Things the runtime image does not need ────────────────────────────────── +# NOTE: migrations/ is deliberately NOT ignored. They are baked into the image +# so a Kubernetes initContainer can apply them using the same image and tag as +# the API — see Dockerfile.api. docker-compose uses a bind mount instead, which +# works either way. +docs/ +scripts/ +infrastructure/ +Makefile +README.md +*.md + +# The snapshot taken before the krowdb public schema was dropped. Large, and +# has no business in a container. +krowdb_public_snapshot_*.sql + +# ── Editor / OS ───────────────────────────────────────────────────────────── +.DS_Store +.idea/ +.vscode/ diff --git a/.gitignore b/.gitignore index 1e1ceee..0c69f65 100644 --- a/.gitignore +++ b/.gitignore @@ -1,8 +1,18 @@ # Secrets and local configuration +# A bare pattern matches at any depth, so this covers infrastructure/.env too. .env .env.local .env.*.local +# TLS material. The local-db overlay generates a self-signed pair inside the +# postgres volume, but nothing stops someone dropping certs here by hand. +*.pem +*.key +*.crt + +# Structural snapshots taken before destructive migrations +krowdb_public_snapshot_*.sql + # Go build output /go-api/bin/ *.test @@ -12,3 +22,6 @@ .DS_Store .idea/ .vscode/ + +# Filled-in Kubernetes secret (the .example is the committed template) +infrastructure/k8s/10-secret.yaml diff --git a/Makefile b/Makefile index 59dad1a..621c6ff 100644 --- a/Makefile +++ b/Makefile @@ -135,3 +135,57 @@ migrate-force: ## Clear a dirty state: make migrate-force VERSION=1 .PHONY: verify-schema verify-schema: ## Print the tables, enums, FKs and indexes that exist in the target schema @$(PSQL) -f scripts/verify_schema.sql + +# ── Docker ────────────────────────────────────────────────────────────────── +# +# These read infrastructure/.env, NOT the repository-root .env that `make run` +# uses. Two files on purpose: the root one points at a developer's database +# from the host, the infrastructure one points at a deployment's database from +# inside a container, and DATABASE_HOST is different in each. + +COMPOSE_DIR := infrastructure +COMPOSE_FILES := -f docker-compose.yml +# Layer the local database in with: make docker-up LOCAL_DB=1 +ifdef LOCAL_DB +COMPOSE_FILES += -f docker-compose.local-db.yml +endif +COMPOSE = cd $(COMPOSE_DIR) && docker compose $(COMPOSE_FILES) + +.PHONY: docker-build +docker-build: ## Build the API image + $(COMPOSE) build + +.PHONY: docker-up +docker-up: ## Start the stack (migrations run first). LOCAL_DB=1 adds PostgreSQL + $(COMPOSE) up -d --build + @echo "api: http://127.0.0.1:8080/health" + @echo "logs: make docker-logs" + +.PHONY: docker-down +docker-down: ## Stop the stack, keeping volumes + $(COMPOSE) down + +.PHONY: docker-logs +docker-logs: ## Follow the API logs + $(COMPOSE) logs -f api + +.PHONY: docker-ps +docker-ps: ## Show stack status and health + $(COMPOSE) ps + +.PHONY: docker-migrate +docker-migrate: ## Apply migrations only, without starting the API + $(COMPOSE) run --rm migrate + +.PHONY: docker-seed +docker-seed: ## Load the demo fixture into the deployment's database + $(COMPOSE) run --rm --entrypoint /usr/local/bin/seed api + +.PHONY: docker-setpassword +docker-setpassword: ## Set a password: make docker-setpassword EMAIL=demo@krow.app + @test -n "$(EMAIL)" || { echo "usage: make docker-setpassword EMAIL=demo@krow.app"; exit 1; } + $(COMPOSE) run --rm -it --entrypoint /usr/local/bin/setpassword api -email $(EMAIL) + +.PHONY: docker-config +docker-config: ## Render the fully resolved compose configuration + $(COMPOSE) config diff --git a/go-api/internal/config/config.go b/go-api/internal/config/config.go index 160ddf1..b8d839e 100644 --- a/go-api/internal/config/config.go +++ b/go-api/internal/config/config.go @@ -47,6 +47,27 @@ type HTTPConfig struct { IdleTimeout time.Duration ShutdownTimeout time.Duration + // CookieSameSite is the SameSite attribute on the session cookie: + // "lax" (default), "none" or "strict". + // + // This exists because CORS is only half of what a cross-origin browser call + // needs, and the other half is easy to miss. SameSite is judged on SITE + // (registrable domain), not origin: + // + // app.krow.com → api.krow.com SAME site. Lax sends the cookie. ✓ + // krow.vercel.app → api.krow.com CROSS site. Lax does NOT send it. ✗ + // localhost:5173 → 127.0.0.1:8080 CROSS site — different hosts. ✗ + // + // So a deployment whose frontend is on an unrelated domain gets a perfect + // set of CORS headers and still no session, because the browser never + // attaches the cookie. "none" is the only value that survives that, and it + // requires Secure, which means HTTPS. + // + // Default stays "lax": it is the safe value, it is correct for the + // same-site and same-origin deployments this is normally run as, and it + // gives CSRF protection that "none" gives up. + CookieSameSite string + // CORSOrigins is the exact set of browser origins allowed to call the API. // // It exists for one reason: in local development the Vite dev server is an @@ -137,6 +158,7 @@ func Load() (*Config, error) { IdleTimeout: durationDefault("HTTP_IDLE_TIMEOUT", 60*time.Second), ShutdownTimeout: durationDefault("HTTP_SHUTDOWN_TIMEOUT", 10*time.Second), CORSOrigins: corsOrigins(withDefault("APP_ENV", "development")), + CookieSameSite: strings.ToLower(withDefault("HTTP_COOKIE_SAMESITE", "lax")), }, Seed: SeedConfig{ FixturePath: withDefault("SEED_FIXTURE_PATH", "./seed/fixtures/seed.json"), @@ -194,12 +216,38 @@ func (c *Config) validate() error { if c.AppEnv == "production" && c.DB.SSLMode == "disable" { return fmt.Errorf("DATABASE_SSLMODE=disable is not allowed when APP_ENV=production") } + switch c.HTTP.CookieSameSite { + case "lax", "strict": + case "none": + // SameSite=None without Secure is ignored — and in current browsers, + // rejected outright — so the cookie would simply never be stored. The + // Secure flag is set for every APP_ENV except development, so this is + // the one combination that produces a silently sessionless deployment. + if c.AppEnv == "development" { + return fmt.Errorf("HTTP_COOKIE_SAMESITE=none requires the Secure cookie flag, " + + "which is not set when APP_ENV=development; SameSite=None over plain HTTP " + + "is rejected by browsers") + } + default: + return fmt.Errorf("HTTP_COOKIE_SAMESITE must be lax, none or strict, got %q", + c.HTTP.CookieSameSite) + } for _, origin := range c.HTTP.CORSOrigins { // "*" is rejected rather than quietly honoured. The middleware echoes a // single matched origin, so a wildcard could only ever be a // misunderstanding of what this setting does. + // "*" is not a stricter-than-necessary policy choice — it cannot work + // here at all. Authentication is a cookie, so the API must answer + // Access-Control-Allow-Credentials: true, and every browser REFUSES + // the pairing of that header with Allow-Origin: "*". A deployment + // configured this way would send correct-looking headers and have + // every authenticated call blocked client-side. if origin == "*" { - return fmt.Errorf("HTTP_CORS_ORIGINS must list explicit origins; \"*\" is not accepted") + return fmt.Errorf(`HTTP_CORS_ORIGINS must list explicit origins; "*" cannot be used ` + + `because this API authenticates with a cookie, and browsers reject ` + + `Access-Control-Allow-Origin: "*" together with credentials. ` + + `List each frontend origin, or serve the frontend from the API's own origin ` + + `(then leave this unset and CORS is not involved at all)`) } if !strings.HasPrefix(origin, "http://") && !strings.HasPrefix(origin, "https://") { return fmt.Errorf("HTTP_CORS_ORIGINS entry %q must be a full origin including the scheme", origin) diff --git a/go-api/internal/httpserver/auth.go b/go-api/internal/httpserver/auth.go index 227cb8d..b2f891e 100644 --- a/go-api/internal/httpserver/auth.go +++ b/go-api/internal/httpserver/auth.go @@ -42,6 +42,29 @@ const sessionCookieName = "krow_session" // that never authenticate anything. func (s *Server) secureCookies() bool { return s.cfg.AppEnv != "development" } +// sameSite resolves the configured SameSite mode. +// +// Lax remains the default and the recommendation. "none" exists for the one +// deployment shape that cannot work without it: a frontend on a different +// registrable domain from the API. In that case Lax withholds the cookie on +// every cross-site fetch, so the sign-in succeeds, the Set-Cookie arrives, and +// the next request carries nothing — which reads as a broken session rather +// than as a cookie policy. +// +// An unrecognised value falls back to Lax rather than to None. config.validate +// rejects those before startup, so this is only a belt-and-braces default in +// the safe direction. +func (s *Server) sameSite() http.SameSite { + switch s.cfg.HTTP.CookieSameSite { + case "none": + return http.SameSiteNoneMode + case "strict": + return http.SameSiteStrictMode + default: + return http.SameSiteLaxMode + } +} + // setSessionCookie writes the raw token to the browser. // // This is the only place the raw token is written to a response, and it goes @@ -59,12 +82,14 @@ func (s *Server) setSessionCookie(w http.ResponseWriter, token string, lifetime Path: "/", // HttpOnly: script cannot read it. HttpOnly: true, - // Lax, not Strict and not None. Strict would drop the cookie on any - // cross-site navigation, so following a link into the app would land on - // a login page despite a live session. None would require Secure and - // would send the cookie on cross-site POSTs, which is the CSRF hole Lax - // exists to close. - SameSite: http.SameSiteLaxMode, + // Lax by default, and Strict/None available through + // HTTP_COOKIE_SAMESITE. Strict would drop the cookie on any cross-site + // navigation, so following a link into the app would land on a login + // page despite a live session. None sends it on cross-site requests, + // which is the CSRF hole Lax exists to close — and is nonetheless the + // only workable value when the frontend is on a different registrable + // domain. See Server.sameSite. + SameSite: s.sameSite(), Secure: s.secureCookies(), MaxAge: int(lifetime.Seconds()), }) @@ -82,7 +107,9 @@ func (s *Server) clearSessionCookie(w http.ResponseWriter) { Value: "", Path: "/", HttpOnly: true, - SameSite: http.SameSiteLaxMode, + // Must match the attributes it was set with, SameSite included, or the + // browser treats this as a different cookie and leaves the original. + SameSite: s.sameSite(), Secure: s.secureCookies(), MaxAge: -1, }) diff --git a/go-api/internal/httpserver/cors.go b/go-api/internal/httpserver/cors.go index df5e6ce..4568d42 100644 --- a/go-api/internal/httpserver/cors.go +++ b/go-api/internal/httpserver/cors.go @@ -73,6 +73,19 @@ func cors(origins []string) func(http.Handler) http.Handler { w.Header().Set("Access-Control-Allow-Origin", origin) + // Authentication is a cookie, so the browser will neither send it + // nor expose the response without this. It is set for allowlisted + // origins only, and the origin above is always a specific one — + // the pairing of Allow-Credentials with "*" is rejected outright by + // browsers, which is the second reason this middleware never echoes + // a wildcard. + // + // Both the preflight and the actual response need it: the preflight + // decides whether the browser is willing to SEND the cookie, and the + // actual response decides whether the page may READ the result. + // Setting it here, before the preflight branch, covers both. + w.Header().Set("Access-Control-Allow-Credentials", "true") + if isPreflight(r) { w.Header().Add("Vary", "Access-Control-Request-Method") w.Header().Add("Vary", "Access-Control-Request-Headers") diff --git a/go-api/internal/httpserver/cors_test.go b/go-api/internal/httpserver/cors_test.go index 64913bc..3330237 100644 --- a/go-api/internal/httpserver/cors_test.go +++ b/go-api/internal/httpserver/cors_test.go @@ -133,6 +133,62 @@ func TestCORSOffByDefault(t *testing.T) { } } +// Authentication is a cookie, so a cross-origin frontend calling with +// `credentials: 'include'` needs Access-Control-Allow-Credentials on the actual +// response. Without it the browser blocks the page from reading a reply the +// server answered perfectly well, and the app sees an opaque network failure +// beside a 200 in the server log. +func TestCORSAllowsCredentialsOnResponse(t *testing.T) { + handler := corsAPI(t, devOrigin) + + rec := send(handler, "GET", "/api/v1/job-postings", map[string]string{"Origin": devOrigin}) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if got := rec.Header().Get("Access-Control-Allow-Credentials"); got != "true" { + t.Fatalf("Access-Control-Allow-Credentials = %q, want \"true\" — "+ + "a cookie-authenticated API is unreadable cross-origin without it", got) + } + // Allow-Credentials with a wildcard origin is rejected by every browser, so + // the two must never appear together. + if got := rec.Header().Get("Access-Control-Allow-Origin"); got == "*" { + t.Fatal("Access-Control-Allow-Origin is \"*\" alongside credentials; browsers refuse that pairing") + } +} + +// The preflight decides whether the browser is willing to SEND the cookie at +// all, so it needs the header too — separately from the actual response. +func TestCORSAllowsCredentialsOnPreflight(t *testing.T) { + handler := corsAPI(t, devOrigin) + + rec := send(handler, "OPTIONS", "/api/v1/job-postings", map[string]string{ + "Origin": devOrigin, + "Access-Control-Request-Method": "POST", + }) + if rec.Code != http.StatusNoContent { + t.Fatalf("expected 204, got %d", rec.Code) + } + if got := rec.Header().Get("Access-Control-Allow-Credentials"); got != "true" { + t.Fatalf("preflight Access-Control-Allow-Credentials = %q, want \"true\"", got) + } +} + +// An origin that is not on the allowlist must not be handed credentials +// permission — the header is worthless on its own, but pairing it with a +// reflected origin would be the classic misconfiguration. +func TestCORSWithholdsCredentialsFromUnknownOrigin(t *testing.T) { + handler := corsAPI(t, devOrigin) + + rec := send(handler, "GET", "/api/v1/job-postings", + map[string]string{"Origin": "http://evil.example"}) + if got := rec.Header().Get("Access-Control-Allow-Credentials"); got != "" { + t.Fatalf("an unlisted origin was granted credentials: %q", got) + } + if got := rec.Header().Get("Access-Control-Allow-Origin"); got != "" { + t.Fatalf("an unlisted origin was echoed back: %q", got) + } +} + func contains(haystack, needle string) bool { for i := 0; i+len(needle) <= len(haystack); i++ { if haystack[i:i+len(needle)] == needle { diff --git a/go-api/internal/httpserver/server.go b/go-api/internal/httpserver/server.go index 59e07a6..937653b 100644 --- a/go-api/internal/httpserver/server.go +++ b/go-api/internal/httpserver/server.go @@ -37,6 +37,7 @@ type Server struct { db *db.DB api *service.Registry definitions *service.DefinitionsService + workflows *service.WorkflowService log *slog.Logger http *http.Server started time.Time @@ -124,6 +125,7 @@ func New(cfg *config.Config, database *db.DB, log *slog.Logger, opts ...Option) cfg: cfg, db: database, log: log, api: service.NewRegistry(database.Pool), definitions: service.NewDefinitions(database.Pool), + workflows: service.NewWorkflows(database.Pool).WithClock(o.now), started: o.now(), sessions: sessions, users: users, @@ -135,7 +137,8 @@ func New(cfg *config.Config, database *db.DB, log *slog.Logger, opts ...Option) mux := http.NewServeMux() mux.HandleFunc("GET /health", s.handleHealth) - s.endpoints = s.routeAuth(mux) + s.routeResources(mux) + s.routeMe(mux) + s.routeDefinitions(mux) + s.endpoints = s.routeAuth(mux) + s.routeResources(mux) + s.routeMe(mux) + + s.routeDefinitions(mux) + s.routeWorkflows(mux) handler := jsonErrors(mux) // Authentication sits where devOrgMiddleware used to, so every route below diff --git a/go-api/internal/httpserver/workflows.go b/go-api/internal/httpserver/workflows.go new file mode 100644 index 0000000..340f49b --- /dev/null +++ b/go-api/internal/httpserver/workflows.go @@ -0,0 +1,153 @@ +package httpserver + +import ( + "net/http" + + "github.com/krow/krow-backend/go-api/internal/authctx" + "github.com/krow/krow-backend/go-api/internal/domain" + "github.com/krow/krow-backend/go-api/internal/service" +) + +// The multi-record endpoints from api-contract.md §12.1. +// +// These are the first routes that are not a plain CRUD projection of a table, +// and they are shaped as verbs on the record they act on — `.../{id}/hire`, +// `.../{id}/assignments` — rather than as new collections. The action is the +// thing being requested, and it has no independent existence to GET. +// +// AUTHORIZATION REUSES THE POLICY TABLE RATHER THAN ADDING TO IT. +// +// A workflow is exactly as privileged as the writes it performs, so each one +// is gated on the operations it will actually carry out — hire needs UPDATE on +// job-applications and CREATE on staff; assign needs CREATE on assignments and +// UPDATE on job-applications. Inventing a separate `hire` permission would +// create a second place where the answer to "who may do this" lives, and the +// two would eventually disagree. Every pair below resolves to `operators` +// today, which is the intended answer: a talent user cannot hire themselves or +// place themselves on a shift. + +// requirement is one (resource, operation) pair a workflow depends on. +type requirement struct { + path string + op domain.Op +} + +// authorizeAll refuses unless the caller may perform every listed operation. +// +// All-or-nothing, checked before any transaction opens: a caller who may update +// an application but not create staff must not get halfway through a hire and +// be rolled back. The refusal is the same 403 a single-operation handler gives, +// and names no resource — see domain.Forbidden. +func (s *Server) authorizeAll(w http.ResponseWriter, r *http.Request, + reqs ...requirement) (authctx.Identity, bool) { + + ident, err := authctx.MustFrom(r.Context()) + if err != nil { + // Unreachable: the middleware refuses an unauthenticated request before + // the router sees it. A missing identity here is a wiring bug. + writeError(w, s.log, domain.Internal(err)) + return authctx.Identity{}, false + } + + role, known := domain.ParseRole(ident.Role) + if !known { + s.log.Warn("workflow refused: unknown role", + "user_id", ident.UserID, "role", ident.Role, "path", r.URL.Path) + writeError(w, s.log, domain.Forbidden()) + return authctx.Identity{}, false + } + + for _, req := range reqs { + svc, ok := s.api.Get(req.path) + if !ok { + writeError(w, s.log, domain.Internal( + errUnregisteredResource(req.path))) + return authctx.Identity{}, false + } + if !svc.Resource().Policy.Allows(req.op, role) { + s.log.Warn("workflow authorization refused", + "user_id", ident.UserID, "role", ident.Role, + "required_resource", req.path, "path", r.URL.Path) + writeError(w, s.log, domain.Forbidden()) + return authctx.Identity{}, false + } + } + return ident, true +} + +type unregisteredResourceError string + +func (e unregisteredResourceError) Error() string { + return "httpserver: workflow depends on unregistered resource " + string(e) +} + +func errUnregisteredResource(path string) error { return unregisteredResourceError(path) } + +func (s *Server) routeWorkflows(mux *http.ServeMux) int { + mux.HandleFunc("POST /api/v1/job-applications/{id}/hire", s.handleHire) + mux.HandleFunc("POST /api/v1/job-postings/{id}/assignments", s.handleAssign) + return 2 +} + +// handleHire moves an application to `hired` and creates the staff record in +// one transaction. Replaces the two-call sequence at krowHooks.js:302-303. +func (s *Server) handleHire(w http.ResponseWriter, r *http.Request) { + ident, ok := s.authorizeAll(w, r, + requirement{"job-applications", domain.OpUpdate}, + requirement{"staff", domain.OpCreate}, + requirement{"user-activity", domain.OpCreate}, + ) + if !ok { + return + } + + body, err := decodeBody(r) + if err != nil { + writeError(w, s.log, err) + return + } + + result, err := s.workflows.Hire(r.Context(), ident, r.PathValue("id"), body) + if err != nil { + writeError(w, s.log, err) + return + } + + s.log.Info("candidate hired", "user_id", ident.UserID, + "application_id", r.PathValue("id"), "staff_id", result.Staff["id"]) + + // 201: the request created a staff record. The application it also updated + // is returned alongside so the caller can render the new state without a + // second read. + writeJSON(w, http.StatusCreated, envelope{Data: result}) +} + +// handleAssign places workers on a posting in one transaction. Replaces the 3n +// sequential round-trips at krowHooks.js:421/449/466. +func (s *Server) handleAssign(w http.ResponseWriter, r *http.Request) { + ident, ok := s.authorizeAll(w, r, + requirement{"assignments", domain.OpCreate}, + requirement{"job-applications", domain.OpUpdate}, + requirement{"user-activity", domain.OpCreate}, + ) + if !ok { + return + } + + var req service.AssignRequest + if err := decodeInto(r, &req); err != nil { + writeError(w, s.log, err) + return + } + + result, err := s.workflows.Assign(r.Context(), ident, r.PathValue("id"), req) + if err != nil { + writeError(w, s.log, err) + return + } + + s.log.Info("workers assigned", "user_id", ident.UserID, + "job_posting_id", r.PathValue("id"), "count", result.Count) + + writeJSON(w, http.StatusCreated, envelope{Data: result}) +} diff --git a/go-api/internal/httpserver/workflows_test.go b/go-api/internal/httpserver/workflows_test.go new file mode 100644 index 0000000..80d3adf --- /dev/null +++ b/go-api/internal/httpserver/workflows_test.go @@ -0,0 +1,331 @@ +package httpserver_test + +import ( + "context" + "net/http" + "testing" +) + +// The multi-record endpoints from api-contract.md §12.1. +// +// The property worth testing here is not that the happy path works — it is that +// a failure part-way through leaves NOTHING behind. Every rollback test below +// counts rows before and after, because "the request returned an error" and +// "the request changed nothing" are different claims and only the second one is +// what a transaction is for. + +// applicationFor creates an application that can be hired. +func applicationFor(t *testing.T, r *rbac, posting, name, email string) string { + t.Helper() + return mustCreate(t, r, r.admin, "/api/v1/job-applications", map[string]any{ + "job_posting_id": posting, + "applicant_name": name, + "email": email, + "status": "shortlisted", + "ai_score": 77, + }) +} + +func countRows(t *testing.T, r *rbac, table string) int { + t.Helper() + var n int + if err := r.h.Pool.QueryRow(context.Background(), + `SELECT count(*) FROM `+table+` WHERE org_id = $1::uuid`, r.orgID).Scan(&n); err != nil { + t.Fatalf("count %s: %v", table, err) + } + return n +} + +/* ── Hire ───────────────────────────────────────────────────────────────── */ + +func TestHireCreatesStaffAndMovesApplication(t *testing.T) { + r := newRBAC(t) + app := applicationFor(t, r, r.activePosting, "Hire Me", "hire-me@example.test") + + before := countRows(t, r, "staff") + got := r.as(r.admin, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{ + "role": "Event Server", "profile_tier": "Skilled", + }) + if got.code != http.StatusCreated { + t.Fatalf("hire: got %d, want 201 (%v)", got.code, got.body) + } + + data, _ := got.body["data"].(map[string]any) + application, _ := data["application"].(map[string]any) + staff, _ := data["staff"].(map[string]any) + if application == nil || staff == nil { + t.Fatalf("hire response is missing application or staff: %v", got.body) + } + + if application["status"] != "hired" { + t.Errorf("application.status = %v, want hired", application["status"]) + } + if staff["name"] != "Hire Me" { + t.Errorf("staff.name = %v, want the applicant's name", staff["name"]) + } + if staff["email"] != "hire-me@example.test" { + t.Errorf("staff.email = %v, want the application's email", staff["email"]) + } + // The caller's overrides win over the derived defaults. + if staff["role"] != "Event Server" { + t.Errorf("staff.role = %v, want the supplied role", staff["role"]) + } + if staff["profile_tier"] != "Skilled" { + t.Errorf("staff.profile_tier = %v, want the supplied tier", staff["profile_tier"]) + } + // ai_score is carried across from the application. It is an `int` column, + // so it arrives from pgx as int32 — the case repo.bindValue did not handle + // until this endpoint existed to read a record and write it elsewhere. + if score, ok := staff["ai_score"].(float64); !ok || int(score) != 77 { + t.Errorf("staff.ai_score = %v, want 77 carried from the application", staff["ai_score"]) + } + if staff["application_id"] != app { + t.Errorf("staff.application_id = %v, want %s", staff["application_id"], app) + } + if after := countRows(t, r, "staff"); after != before+1 { + t.Errorf("staff rows: %d -> %d, want exactly one more", before, after) + } +} + +// Hiring the same application twice would create a second employment record for +// one person, so the second attempt is a conflict rather than a repeat. +func TestHireIsNotRepeatable(t *testing.T) { + r := newRBAC(t) + app := applicationFor(t, r, r.activePosting, "Twice", "twice@example.test") + + if got := r.as(r.admin, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{}); got.code != http.StatusCreated { + t.Fatalf("first hire: got %d, want 201 (%v)", got.code, got.body) + } + + before := countRows(t, r, "staff") + got := r.as(r.admin, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{}) + if got.code != http.StatusConflict { + t.Fatalf("second hire: got %d, want 409 (%v)", got.code, got.body) + } + if after := countRows(t, r, "staff"); after != before { + t.Errorf("a refused hire still wrote a staff row: %d -> %d", before, after) + } +} + +// The whole point of the endpoint: the two writes succeed together or not at +// all. A staff insert that violates a constraint must leave the application +// untouched, not merely report an error. +func TestHireRollsBackTheApplicationWhenStaffFails(t *testing.T) { + r := newRBAC(t) + app := applicationFor(t, r, r.activePosting, "Rollback", "rollback@example.test") + + staffBefore := countRows(t, r, "staff") + + // profile_tier is a native enum; a value outside it fails the staff INSERT + // after the application UPDATE has already been issued in this transaction. + got := r.as(r.admin, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{ + "profile_tier": "NotARealTier", + }) + if got.code == http.StatusCreated { + t.Fatalf("an invalid profile_tier was accepted: %v", got.body) + } + + if after := countRows(t, r, "staff"); after != staffBefore { + t.Errorf("staff rows changed despite a failed hire: %d -> %d", staffBefore, after) + } + + // The decisive assertion: the application must NOT be hired. + reread := r.as(r.admin, "GET", "/api/v1/job-applications?limit=500", nil) + if reread.code != http.StatusOK { + t.Fatalf("re-read applications: %d", reread.code) + } + for _, raw := range reread.body["data"].([]any) { + rec := raw.(map[string]any) + if rec["id"] == app && rec["status"] == "hired" { + t.Fatal("the application was left hired after the staff insert failed — " + + "the two writes are not in one transaction") + } + } +} + +// Hiring is an operator action. A talent user must not be able to hire anyone, +// including themselves. +func TestHireIsRefusedToTalent(t *testing.T) { + r := newRBAC(t) + app := applicationFor(t, r, r.activePosting, "Self", r.talA.email) + + before := countRows(t, r, "staff") + got := r.as(r.talA, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{}) + if got.code != http.StatusForbidden { + t.Fatalf("talent hire: got %d, want 403 (%v)", got.code, got.body) + } + if after := countRows(t, r, "staff"); after != before { + t.Errorf("a refused hire still wrote a staff row: %d -> %d", before, after) + } +} + +func TestHireRejectsUnknownApplication(t *testing.T) { + r := newRBAC(t) + got := r.as(r.admin, "POST", + "/api/v1/job-applications/00000000-0000-0000-0000-000000000000/hire", map[string]any{}) + if got.code != http.StatusNotFound { + t.Fatalf("hire unknown application: got %d, want 404 (%v)", got.code, got.body) + } +} + +// Another organization's application is absent, not forbidden — the same 404 a +// nonexistent id gets, so existence does not leak across tenants. +func TestHireCannotReachAnotherOrganization(t *testing.T) { + r := newRBAC(t) + app := applicationFor(t, r, r.activePosting, "Ours", "ours@example.test") + + got := r.as(r.outsider, "POST", "/api/v1/job-applications/"+app+"/hire", map[string]any{}) + if got.code != http.StatusNotFound { + t.Fatalf("cross-tenant hire: got %d, want 404 (%v)", got.code, got.body) + } +} + +/* ── Assign ─────────────────────────────────────────────────────────────── */ + +func TestAssignPlacesWorkersAndUpdatesApplications(t *testing.T) { + r := newRBAC(t) + a1 := applicationFor(t, r, r.activePosting, "Worker One", "w1@example.test") + a2 := applicationFor(t, r, r.activePosting, "Worker Two", "w2@example.test") + + before := countRows(t, r, "assignments") + got := r.as(r.admin, "POST", "/api/v1/job-postings/"+r.activePosting+"/assignments", map[string]any{ + "workers": []map[string]any{ + {"worker_email": "w1@example.test", "worker_name": "Worker One", + "starts_at": "2026-09-01T09:00:00Z", "application_id": a1, "match_score": 91}, + {"worker_email": "w2@example.test", "worker_name": "Worker Two", + "starts_at": "2026-09-01T09:00:00Z", "application_id": a2}, + }, + }) + if got.code != http.StatusCreated { + t.Fatalf("assign: got %d, want 201 (%v)", got.code, got.body) + } + + data, _ := got.body["data"].(map[string]any) + if count, ok := data["count"].(float64); !ok || int(count) != 2 { + t.Errorf("count = %v, want 2", data["count"]) + } + if after := countRows(t, r, "assignments"); after != before+2 { + t.Errorf("assignment rows: %d -> %d, want two more", before, after) + } + + // Both applications must now read as assigned. + list := r.as(r.admin, "GET", "/api/v1/job-applications?status=assigned", nil) + assigned := map[string]bool{} + for _, raw := range list.body["data"].([]any) { + assigned[raw.(map[string]any)["id"].(string)] = true + } + if !assigned[a1] || !assigned[a2] { + t.Errorf("applications were not moved to assigned: a1=%v a2=%v", assigned[a1], assigned[a2]) + } +} + +// A worker with no application is legitimate — that is what the talent pool is +// for — and must not be invented one. +func TestAssignAcceptsWorkerWithoutApplication(t *testing.T) { + r := newRBAC(t) + got := r.as(r.admin, "POST", "/api/v1/job-postings/"+r.activePosting+"/assignments", map[string]any{ + "workers": []map[string]any{ + {"worker_email": "pool@example.test", "worker_name": "Pool Worker", + "starts_at": "2026-09-01T09:00:00Z"}, + }, + }) + if got.code != http.StatusCreated { + t.Fatalf("assign without application: got %d, want 201 (%v)", got.code, got.body) + } +} + +// The batch is all-or-nothing. A bad reference on the SECOND worker must undo +// the first worker's assignment, not leave it stranded. +func TestAssignRollsBackTheWholeBatch(t *testing.T) { + r := newRBAC(t) + before := countRows(t, r, "assignments") + + got := r.as(r.admin, "POST", "/api/v1/job-postings/"+r.activePosting+"/assignments", map[string]any{ + "workers": []map[string]any{ + {"worker_email": "first@example.test", "worker_name": "First", + "starts_at": "2026-09-01T09:00:00Z"}, + {"worker_email": "second@example.test", "worker_name": "Second", + "starts_at": "2026-09-01T09:00:00Z", + "application_id": "00000000-0000-0000-0000-000000000000"}, + }, + }) + if got.code == http.StatusCreated { + t.Fatalf("a batch naming a nonexistent application was accepted: %v", got.body) + } + if after := countRows(t, r, "assignments"); after != before { + t.Fatalf("the first worker survived the second's failure: %d -> %d — "+ + "the batch is not one transaction", before, after) + } +} + +func TestAssignIsRefusedToTalent(t *testing.T) { + r := newRBAC(t) + before := countRows(t, r, "assignments") + + got := r.as(r.talA, "POST", "/api/v1/job-postings/"+r.activePosting+"/assignments", map[string]any{ + "workers": []map[string]any{ + {"worker_email": r.talA.email, "worker_name": "Self", + "starts_at": "2026-09-01T09:00:00Z"}, + }, + }) + if got.code != http.StatusForbidden { + t.Fatalf("talent assign: got %d, want 403 (%v)", got.code, got.body) + } + if after := countRows(t, r, "assignments"); after != before { + t.Errorf("a refused assign still wrote a row: %d -> %d", before, after) + } +} + +func TestAssignValidatesTheBatchBeforeWriting(t *testing.T) { + r := newRBAC(t) + before := countRows(t, r, "assignments") + + cases := []struct { + name string + body map[string]any + }{ + {"no workers", map[string]any{"workers": []map[string]any{}}}, + {"missing email", map[string]any{"workers": []map[string]any{ + {"worker_name": "No Email", "starts_at": "2026-09-01T09:00:00Z"}}}}, + {"missing starts_at", map[string]any{"workers": []map[string]any{ + {"worker_email": "x@example.test", "worker_name": "No Start"}}}}, + {"malformed application_id", map[string]any{"workers": []map[string]any{ + {"worker_email": "x@example.test", "starts_at": "2026-09-01T09:00:00Z", + "application_id": "not-a-uuid"}}}}, + } + for _, tc := range cases { + got := r.as(r.admin, "POST", "/api/v1/job-postings/"+r.activePosting+"/assignments", tc.body) + if got.code != http.StatusUnprocessableEntity { + t.Errorf("%s: got %d, want 422 (%v)", tc.name, got.code, got.body) + } + } + if after := countRows(t, r, "assignments"); after != before { + t.Errorf("a rejected batch wrote rows: %d -> %d", before, after) + } +} + +func TestAssignRejectsUnknownPosting(t *testing.T) { + r := newRBAC(t) + got := r.as(r.admin, "POST", + "/api/v1/job-postings/00000000-0000-0000-0000-000000000000/assignments", map[string]any{ + "workers": []map[string]any{ + {"worker_email": "x@example.test", "starts_at": "2026-09-01T09:00:00Z"}, + }, + }) + if got.code != http.StatusNotFound { + t.Fatalf("assign to unknown posting: got %d, want 404 (%v)", got.code, got.body) + } +} + +// Both endpoints are behind the session like everything else. +func TestWorkflowEndpointsRequireASession(t *testing.T) { + r := newRBAC(t) + for _, path := range []string{ + "/api/v1/job-applications/00000000-0000-0000-0000-000000000000/hire", + "/api/v1/job-postings/00000000-0000-0000-0000-000000000000/assignments", + } { + if got := r.doAnon("POST", path, map[string]any{}); got.code != http.StatusUnauthorized { + t.Errorf("%s unauthenticated: got %d, want 401", path, got.code) + } + } +} diff --git a/go-api/internal/repo/repo.go b/go-api/internal/repo/repo.go index f758649..8df3084 100644 --- a/go-api/internal/repo/repo.go +++ b/go-api/internal/repo/repo.go @@ -572,6 +572,13 @@ func bindValue(c domain.Column, v any) (any, error) { return raw, nil case domain.KindInt: + // The int8/16/32 cases are not JSON shapes — encoding/json only ever + // produces float64. They are the widths pgx hands back when a value was + // READ from the database and is being written somewhere else: an `int` + // column arrives as int32, and a service that copies a field from one + // record to another (see service.Hire, which carries ai_score from an + // application onto the staff row) would otherwise be told its own + // database's value "must be a number". switch t := v.(type) { case float64: if t != float64(int64(t)) { @@ -592,19 +599,45 @@ func bindValue(c domain.Column, v any) (any, error) { return t, nil case int: return int64(t), nil + case int32: + return int64(t), nil + case int16: + return int64(t), nil + case int8: + return int64(t), nil + case float32: + if t != float32(int64(t)) { + return nil, domain.Validation( + fmt.Sprintf("%s must be a whole number", c.Name), + map[string]string{c.Name: "expected an integer"}) + } + return int64(t), nil } return nil, domain.Validation(fmt.Sprintf("%s must be a number", c.Name), nil) case domain.KindFloat: + // float32 and the integer widths for the same reason as above: a + // numeric column is projected as float8 and returns float64, but an + // integer read from elsewhere may legitimately be written into one. switch t := v.(type) { case float64: return t, nil + case float32: + return float64(t), nil case string: f, err := strconv.ParseFloat(t, 64) if err != nil { return nil, domain.Validation(fmt.Sprintf("%s must be a number", c.Name), nil) } return f, nil + case int64: + return float64(t), nil + case int: + return float64(t), nil + case int32: + return float64(t), nil + case int16: + return float64(t), nil } return nil, domain.Validation(fmt.Sprintf("%s must be a number", c.Name), nil) diff --git a/go-api/internal/service/workflows.go b/go-api/internal/service/workflows.go new file mode 100644 index 0000000..2076aa2 --- /dev/null +++ b/go-api/internal/service/workflows.go @@ -0,0 +1,443 @@ +package service + +// Multi-record writes, in one transaction each. +// +// api-contract.md §12.1 lists four flows that the frontend performs as a +// sequence of independent HTTP calls, with no transaction and no rollback: +// hiring a candidate, assigning workers to a posting, screening a whole list, +// and submitting a challenge. A failure halfway through leaves the database +// inconsistent — an application marked hired with no staff row, or an +// assignment with an application still showing as merely shortlisted. +// +// This file collapses the first two into one endpoint and one transaction +// each, which is what §12.1 says Phase 3 should do. The other two are not here: +// `useScreenAllCandidates` is n independent PATCHes that are individually +// meaningful (a partial screen is not a corrupt state), and `useSubmitChallenge` +// writes evidence the talent user owns, which needs the talent-scoped predicate +// and is a different shape of problem. +// +// WHY THE REPOSITORY IS REUSED RATHER THAN HAND-WRITTEN SQL +// +// Every write below goes through repo.Repo, built over the transaction rather +// than the pool. That is deliberate: the repository is where org_id is forced +// from the session, where server-derived columns are filled, where values are +// bound and cast to their declared types, and where a constraint violation is +// translated into the contract's error codes. Writing raw SQL here would mean +// re-deriving all four, and getting one of them subtly wrong. + +import ( + "context" + "errors" + "fmt" + "time" + + "github.com/jackc/pgx/v5" + + "github.com/krow/krow-backend/go-api/internal/authctx" + "github.com/krow/krow-backend/go-api/internal/domain" + "github.com/krow/krow-backend/go-api/internal/repo" +) + +// maxAssignmentBatch bounds one assign call. +// +// `useAssignWorkers` assigns a selection from the talent pool, which is a +// human-sized list. The cap exists so a single request cannot hold a +// transaction open across thousands of inserts, blocking every other write to +// these tables for the duration. +const maxAssignmentBatch = 200 + +// TxBeginner is the part of the pool this file needs. An interface rather than +// *pgxpool.Pool so the service stays testable and consistent with repo.Querier. +type TxBeginner interface { + Begin(ctx context.Context) (pgx.Tx, error) +} + +// WorkflowService owns the multi-record flows. +type WorkflowService struct { + db TxBeginner + // now is injectable so tests can assert on generated dates without + // depending on the day they run. + now func() time.Time +} + +// NewWorkflows builds the workflow service over a pool. +func NewWorkflows(db TxBeginner) *WorkflowService { + return &WorkflowService{db: db, now: time.Now} +} + +// WithClock replaces the clock. For tests. +func (s *WorkflowService) WithClock(now func() time.Time) *WorkflowService { + if now != nil { + s.now = now + } + return s +} + +// inTx runs fn inside a transaction, rolling back on any error. +// +// The rollback is deferred rather than called on each error path: an early +// return, a panic in a callee, and an explicit failure all have to undo the +// work, and only a deferred rollback covers the second. Rolling back an +// already-committed transaction is a no-op in pgx, so the deferred call is safe +// on the success path too. +func (s *WorkflowService) inTx(ctx context.Context, fn func(tx pgx.Tx) error) error { + tx, err := s.db.Begin(ctx) + if err != nil { + return fmt.Errorf("begin transaction: %w", err) + } + defer func() { _ = tx.Rollback(ctx) }() + + if err := fn(tx); err != nil { + return err + } + return tx.Commit(ctx) +} + +func resourceByPath(path string) (*domain.Resource, error) { + res, ok := domain.ResourceByPath[path] + if !ok { + return nil, domain.Internal(fmt.Errorf("service: resource %q is not registered", path)) + } + return res, nil +} + +/* ── Hire ───────────────────────────────────────────────────────────────── */ + +// HireResult is what a completed hire returns: both records the flow touched, +// so the caller does not need a follow-up read to render the outcome. +type HireResult struct { + Application domain.Record `json:"application"` + Staff domain.Record `json:"staff"` +} + +// Hire moves an application to `hired` and creates the staff record, atomically. +// +// Replaces the two-call sequence at krowHooks.js:302-303. The failure that +// motivated it: the PATCH succeeds, the POST fails, and the candidate is now +// hired with no employment record and no way for the UI to notice. +// +// Fields the caller may supply are the ones a hiring form collects — hire_date, +// role, phone, profile_tier, status, reviewer_name. Everything else is carried +// across from the application, because it is already the truth about this +// person and retyping it is how the two records drift apart. +func (s *WorkflowService) Hire(ctx context.Context, ident authctx.Identity, + applicationID string, body domain.Record) (*HireResult, error) { + + if !isUUID(applicationID) { + return nil, domain.NotFound("JobApplication", applicationID) + } + + apps, err := resourceByPath("job-applications") + if err != nil { + return nil, err + } + staffRes, err := resourceByPath("staff") + if err != nil { + return nil, err + } + activity, err := resourceByPath("user-activity") + if err != nil { + return nil, err + } + + var out HireResult + err = s.inTx(ctx, func(tx pgx.Tx) error { + appRepo := repo.New(apps, tx) + + // Read inside the transaction. Reading outside it would leave a window + // where two concurrent hires both see a not-yet-hired application. + app, err := appRepo.Get(ctx, ident, applicationID) + if err != nil { + return err + } + if app == nil { + return domain.NotFound("JobApplication", applicationID) + } + + // Hiring twice is a conflict, not an idempotent repeat: the second call + // would create a second employment record for one application. 409 + // rather than 200 because the caller asked for something that cannot be + // done, and silently returning the first hire would hide a double + // submission rather than report it. + if status, _ := app["status"].(string); status == "hired" { + return domain.Conflict("this application has already been hired") + } + + email, _ := app["email"].(string) + name, _ := app["applicant_name"].(string) + if email == "" || name == "" { + return domain.Validation( + "this application cannot be hired: it has no applicant name or email", + map[string]string{"application": "incomplete"}) + } + + staffRecord := domain.Record{ + "application_id": applicationID, + "job_posting_id": app["job_posting_id"], + "worker_profile_id": app["worker_profile_id"], + "name": name, + "email": email, + "phone": pick(body, "phone", app["phone"]), + "role": pick(body, "role", app["job_title"]), + "hire_date": pick(body, "hire_date", s.now().UTC().Format("2006-01-02")), + "ai_score": app["ai_score"], + "profile_tier": pick(body, "profile_tier", "Beginner"), + "status": pick(body, "status", "onboarding"), + "client_rating": app["client_rating"], + "reviewer_name": pick(body, "reviewer_name", ""), + } + // A null worker_profile_id is legitimate — not every applicant has a + // profile — but the column list must not carry an explicit nil for a + // NOT NULL column, so drop the keys the application had nothing for. + dropNil(staffRecord, "worker_profile_id", "job_posting_id") + + hired, err := appRepo.Update(ctx, ident, applicationID, domain.Record{"status": "hired"}) + if err != nil { + return err + } + if hired == nil { + // Unreachable: Get above found it under the same predicate. + return domain.NotFound("JobApplication", applicationID) + } + + created, err := repo.New(staffRes, tx).Insert(ctx, ident, staffRecord) + if err != nil { + return err + } + + // The audit entry is part of the transaction on purpose: a hire that + // happened without a log entry, or a log entry for a hire that rolled + // back, are both worse than neither. + if _, err := repo.New(activity, tx).Insert(ctx, ident, domain.Record{ + "event_type": "candidate_hired", + "details": fmt.Sprintf("%s was hired", name), + "application_id": applicationID, + "position_id": app["job_posting_id"], + "worker_email": email, + }); err != nil { + return err + } + + out.Application, out.Staff = hired, created + return nil + }) + if err != nil { + return nil, err + } + return &out, nil +} + +/* ── Assign ─────────────────────────────────────────────────────────────── */ + +// AssignWorker is one worker in an assign request. +type AssignWorker struct { + WorkerEmail string `json:"worker_email"` + WorkerName string `json:"worker_name"` + StartsAt string `json:"starts_at"` + EndsAt *string `json:"ends_at"` + ApplicationID *string `json:"application_id"` + WorkerProfileID *string `json:"worker_profile_id"` + MatchScore *int `json:"match_score"` + Source string `json:"source"` +} + +// AssignRequest is the body of POST /job-postings/{id}/assignments. +type AssignRequest struct { + Workers []AssignWorker `json:"workers"` +} + +// AssignResult reports what the batch created. +type AssignResult struct { + Assignments []domain.Record `json:"assignments"` + Count int `json:"count"` +} + +// Assign places workers on a posting, atomically. +// +// Replaces the 3n sequential round-trips at krowHooks.js:421/449/466 with one +// request and one transaction. All-or-nothing across the whole batch: assigning +// six workers and having the fourth fail should not leave three assigned, three +// not, and the caller unsure which. +func (s *WorkflowService) Assign(ctx context.Context, ident authctx.Identity, + postingID string, req AssignRequest) (*AssignResult, error) { + + if !isUUID(postingID) { + return nil, domain.NotFound("JobPosting", postingID) + } + if len(req.Workers) == 0 { + return nil, domain.Validation("at least one worker is required", + map[string]string{"workers": "must not be empty"}) + } + if len(req.Workers) > maxAssignmentBatch { + return nil, domain.Validation( + fmt.Sprintf("at most %d workers can be assigned in one request", maxAssignmentBatch), + map[string]string{"workers": "too many"}) + } + + // Validate the whole batch before opening a transaction. A malformed + // request should never have caused a BEGIN. + details := map[string]string{} + for i, w := range req.Workers { + if w.WorkerEmail == "" { + details[fmt.Sprintf("workers[%d].worker_email", i)] = "required" + } + if w.StartsAt == "" { + details[fmt.Sprintf("workers[%d].starts_at", i)] = "required" + } + if w.ApplicationID != nil && !isUUID(*w.ApplicationID) { + details[fmt.Sprintf("workers[%d].application_id", i)] = "must be a uuid" + } + if w.WorkerProfileID != nil && !isUUID(*w.WorkerProfileID) { + details[fmt.Sprintf("workers[%d].worker_profile_id", i)] = "must be a uuid" + } + } + if len(details) > 0 { + return nil, domain.Validation("assignment payload is not valid", details) + } + + postings, err := resourceByPath("job-postings") + if err != nil { + return nil, err + } + assignments, err := resourceByPath("assignments") + if err != nil { + return nil, err + } + apps, err := resourceByPath("job-applications") + if err != nil { + return nil, err + } + activity, err := resourceByPath("user-activity") + if err != nil { + return nil, err + } + + out := &AssignResult{Assignments: []domain.Record{}} + err = s.inTx(ctx, func(tx pgx.Tx) error { + posting, err := repo.New(postings, tx).Get(ctx, ident, postingID) + if err != nil { + return err + } + if posting == nil { + return domain.NotFound("JobPosting", postingID) + } + + assignRepo := repo.New(assignments, tx) + appRepo := repo.New(apps, tx) + activityRepo := repo.New(activity, tx) + + for i, w := range req.Workers { + record := domain.Record{ + "job_posting_id": postingID, + "worker_email": w.WorkerEmail, + "worker_name": w.WorkerName, + "starts_at": w.StartsAt, + "status": "active", + "source": orDefault(w.Source, "manual"), + } + if w.EndsAt != nil { + record["ends_at"] = *w.EndsAt + } + if w.ApplicationID != nil { + record["application_id"] = *w.ApplicationID + } + if w.WorkerProfileID != nil { + record["worker_profile_id"] = *w.WorkerProfileID + } + if w.MatchScore != nil { + record["match_score"] = *w.MatchScore + } + + created, err := assignRepo.Insert(ctx, ident, record) + if err != nil { + return annotate(err, i) + } + out.Assignments = append(out.Assignments, created) + + // The application moves to `assigned` only when one was named. A + // worker can be placed without having applied — that is what the + // talent pool is for — and inventing an application for them would + // be worse than leaving the link absent. + if w.ApplicationID != nil { + updated, err := appRepo.Update(ctx, ident, *w.ApplicationID, + domain.Record{"status": "assigned"}) + if err != nil { + return annotate(err, i) + } + if updated == nil { + return domain.NotFound("JobApplication", *w.ApplicationID) + } + } + + entry := domain.Record{ + "event_type": "worker_assigned", + "details": fmt.Sprintf("%s was assigned", orDefault(w.WorkerName, w.WorkerEmail)), + "position_id": postingID, + "worker_email": w.WorkerEmail, + } + if w.ApplicationID != nil { + entry["application_id"] = *w.ApplicationID + } + if _, err := activityRepo.Insert(ctx, ident, entry); err != nil { + return annotate(err, i) + } + } + return nil + }) + if err != nil { + return nil, err + } + out.Count = len(out.Assignments) + return out, nil +} + +/* ── Helpers ────────────────────────────────────────────────────────────── */ + +// pick takes the caller's value for a key when they supplied a usable one, and +// the fallback otherwise. A present-but-null key means "use the fallback" +// rather than "write null", because these columns are NOT NULL. +func pick(body domain.Record, key string, fallback any) any { + if body == nil { + return fallback + } + v, ok := body[key] + if !ok || v == nil { + return fallback + } + if s, isStr := v.(string); isStr && s == "" { + return fallback + } + return v +} + +// dropNil removes keys whose value is nil, so a NOT NULL column is left to its +// default instead of being sent an explicit null. +func dropNil(rec domain.Record, keys ...string) { + for _, k := range keys { + if v, ok := rec[k]; ok && v == nil { + delete(rec, k) + } + } +} + +func orDefault(v, fallback string) string { + if v == "" { + return fallback + } + return v +} + +// annotate points a validation error at the batch element that produced it. +// Without this, "worker_email must not be null" on a batch of forty says +// nothing about which one. +func annotate(err error, index int) error { + var apiErr *domain.Error + if !errors.As(err, &apiErr) || apiErr.Details == nil { + return err + } + moved := make(map[string]string, len(apiErr.Details)) + for k, v := range apiErr.Details { + moved[fmt.Sprintf("workers[%d].%s", index, k)] = v + } + return domain.Validation(apiErr.Message, moved) +} diff --git a/infrastructure/.env.docker.example b/infrastructure/.env.docker.example new file mode 100644 index 0000000..efde568 --- /dev/null +++ b/infrastructure/.env.docker.example @@ -0,0 +1,95 @@ +# ============================================================================ +# Krow API — Docker environment +# +# cd infrastructure && cp .env.docker.example .env +# +# Compose reads `.env` from the directory the compose file lives in, so this +# copy belongs in infrastructure/, NOT the repository root .env used by +# `make run`. Both are gitignored. +# +# Every value here is a placeholder. No real password or host belongs in a +# file that is committed. +# ============================================================================ + +# ── Application ───────────────────────────────────────────────────────────── +APP_ENV=production # development | staging | production +LOG_LEVEL=info # debug | info | warn | error +VERSION=1.0.0 # build arg only +IMAGE=doormile/krowbackend:latest # what `docker compose build` tags and `up` runs + +# ── Where the API is published ────────────────────────────────────────────── +# Loopback by default: put a reverse proxy in front to terminate TLS. The API +# speaks plain HTTP and its session cookie is Secure outside development, so a +# browser will not send that cookie back over a plain connection anyway. +API_BIND=127.0.0.1 +API_PORT=8080 + +# Browser origins allowed to call this API, comma-separated. Exact strings: +# scheme included, no trailing slash. Cookies are the auth mechanism, so +# anything listed here can hold a live session. +# +# "*" is REJECTED at startup — browsers refuse Allow-Origin: "*" together with +# credentials, so it would break every authenticated call rather than loosen +# anything. +HTTP_CORS_ORIGINS=https://platform.krowforce.com,https://mcp.krowforce.com + +# SameSite on the session cookie: lax | none | strict. +# +# "lax" is right while the API is on *.krowforce.com — the two origins above +# are then cross-ORIGIN (CORS applies) but same-SITE (the cookie is still +# sent). Move the API to any other registrable domain and this must become +# "none", or login will succeed and every following request will arrive +# anonymous. +HTTP_COOKIE_SAMESITE=lax + +# ── Database ──────────────────────────────────────────────────────────────── +# Point at a managed PostgreSQL. With docker-compose.local-db.yml layered on +# top, set DATABASE_HOST=postgres instead. +DATABASE_HOST=your-database-host.example.com +DATABASE_PORT=5432 +DATABASE_NAME=krow +DATABASE_USER=krow +DATABASE_PASSWORD=change-me +DATABASE_SCHEMA=public + +# require encrypts but does not verify the server; verify-full also checks the +# certificate against a CA and is what you want against a managed database. +# +# `disable` is REJECTED at startup when APP_ENV=production. That is deliberate. +DATABASE_SSLMODE=require + +# ── Migration URL ─────────────────────────────────────────────────────────── +# golang-migrate takes a single URL, and Compose cannot assemble or encode one. +# +# ⚠️ PERCENT-ENCODE THE PASSWORD. A password containing @ : / ? # or % will +# otherwise be parsed as part of the host or the path, and the failure looks +# like a wrong host rather than a wrong password: +# +# @ → %40 : → %3A / → %2F ? → %3F # → %23 % → %25 +# +# python3 -c 'import urllib.parse,sys; print(urllib.parse.quote(sys.argv[1], safe=""))' 'your-password' +# +# search_path must match DATABASE_SCHEMA above. +DATABASE_URL=postgres://krow:change-me@your-database-host.example.com:5432/krow?sslmode=require&search_path=public + +# ── Pool and timeouts ─────────────────────────────────────────────────────── +DATABASE_MAX_OPEN_CONNS=25 +DATABASE_MIN_IDLE_CONNS=2 +DATABASE_CONN_MAX_LIFETIME=30m +DATABASE_CONNECT_TIMEOUT=10s +DATABASE_STATEMENT_TIMEOUT=10s + +# ── HTTP timeouts ─────────────────────────────────────────────────────────── +HTTP_READ_TIMEOUT=15s +HTTP_WRITE_TIMEOUT=30s +HTTP_IDLE_TIMEOUT=60s +HTTP_SHUTDOWN_TIMEOUT=10s + +# ── Container resources ───────────────────────────────────────────────────── +API_CPU_LIMIT=2 +API_MEMORY_LIMIT=512M + +# ── Local database only (docker-compose.local-db.yml) ─────────────────────── +POSTGRES_BIND=127.0.0.1 +POSTGRES_PORT=5432 +POSTGRES_MEMORY_LIMIT=1G diff --git a/infrastructure/Dockerfile.api b/infrastructure/Dockerfile.api new file mode 100644 index 0000000..47d45a8 --- /dev/null +++ b/infrastructure/Dockerfile.api @@ -0,0 +1,155 @@ +# syntax=docker/dockerfile:1.7 +# ============================================================================ +# Krow API — production image +# +# Build from the REPOSITORY ROOT, not from this directory: the build needs +# go-api/ and seed/, and a Dockerfile can never reach above its context. +# +# docker build -f infrastructure/Dockerfile.api -t krow-api:latest . +# +# Three binaries ship in the image, because all three are things an operator +# needs against a running deployment and none of them justify a second image: +# +# /usr/local/bin/api the HTTP service (the default command) +# /usr/local/bin/seed loads the demo fixture +# /usr/local/bin/setpassword sets a user's password — without it a fresh +# database has no one who can sign in +# +# Migrations are deliberately NOT run by this image. They are a discrete deploy +# step against the target database BEFORE the new binary rolls out, which is +# what makes expand/migrate/contract possible: see README.md "Staging and +# production". docker-compose.yml runs them as their own one-shot service. +# ============================================================================ + +# ── Build ─────────────────────────────────────────────────────────────────── +# --platform=$BUILDPLATFORM pins this stage to the MACHINE DOING THE BUILDING, +# not the machine that will run the image. Combined with GOARCH below, that +# turns a cross-platform build into cross-COMPILATION rather than emulation. +# +# It matters a lot in practice: building linux/amd64 from an Apple Silicon Mac +# without this runs the entire Go toolchain under QEMU, which is roughly an +# order of magnitude slower. Every binary here is CGO_ENABLED=0 and therefore +# pure Go, so the Go compiler can target another architecture directly and there +# is nothing emulation would buy. +FROM --platform=$BUILDPLATFORM golang:1.27-alpine AS build + +# Supplied automatically by buildx from --platform. Declared here so the build +# stage can read them; TARGETARCH is "amd64", "arm64", etc. +ARG TARGETOS=linux +ARG TARGETARCH + +WORKDIR /src + +# Dependencies first, in their own layer. go.mod and go.sum change far less +# often than source, so an edit to a handler does not re-download the module +# graph. +COPY go-api/go.mod go-api/go.sum ./ +RUN --mount=type=cache,target=/go/pkg/mod \ + go mod download + +COPY go-api/ ./ + +# CGO_ENABLED=0 produces a static binary, which is what lets the final stage be +# a near-empty image with no libc to keep patched. +# -trimpath keeps build-machine paths out of panics and binaries +# -s -w drops the symbol table and DWARF; roughly a third off the size +ARG VERSION=dev +RUN --mount=type=cache,target=/go/pkg/mod \ + --mount=type=cache,target=/root/.cache/go-build \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go build \ + -trimpath -ldflags="-s -w" \ + -o /out/api ./cmd/api && \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go build \ + -trimpath -ldflags="-s -w" -o /out/seed ./cmd/seed && \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go build \ + -trimpath -ldflags="-s -w" -o /out/setpassword ./cmd/setpassword + +# The golang-migrate CLI, built here rather than pulled as a second image. +# +# `go install pkg@version` resolves in its own module context, so this does NOT +# touch go.mod or go.sum. Having it in the image means the Kubernetes manifests +# can run migrations from an initContainer using the SAME image and tag as the +# API — so the schema and the binary that expects it can never be different +# versions, which is the failure a separate migrate image invites. +ARG MIGRATE_VERSION=v4.19.0 +RUN --mount=type=cache,target=/go/pkg/mod \ + --mount=type=cache,target=/root/.cache/go-build \ + CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} go install -tags 'postgres' \ + github.com/golang-migrate/migrate/v4/cmd/migrate@${MIGRATE_VERSION} && \ + cp "$(go env GOPATH)/bin/${TARGETOS}_${TARGETARCH}/migrate" /out/migrate 2>/dev/null || \ + cp "$(go env GOPATH)/bin/migrate" /out/migrate + +# Fail the build rather than the deploy if the tests do not pass. Skipped by +# default because the database-backed tests need PostgreSQL, which a build +# container does not have; turn it on in CI where a service container does. +# Note: this runs on the BUILD architecture, so it is only meaningful when +# building natively. Under cross-compilation the test binaries would target the +# other architecture and could not execute here. +ARG RUN_TESTS=false +RUN if [ "$RUN_TESTS" = "true" ]; then go test ./... ; fi + + +# ── Runtime ───────────────────────────────────────────────────────────────── +# Alpine rather than distroless, for one concrete reason: HEALTHCHECK needs a +# command inside the container, and distroless has no shell and no wget. The +# trade is ~8 MB and an apk surface to keep updated, in exchange for the +# orchestrator being able to see whether this instance is actually serving. +FROM alpine:3.20 AS runtime + +# ca-certificates: the pgx driver needs a trust store to verify a managed +# database's TLS certificate under sslmode=verify-full. +# tzdata: timestamps are timestamptz and the seeder anchors shift dates to now. +RUN apk add --no-cache ca-certificates tzdata wget && \ + adduser -D -H -u 10001 -s /sbin/nologin krow + +COPY --from=build /out/api /usr/local/bin/api +COPY --from=build /out/seed /usr/local/bin/seed +COPY --from=build /out/setpassword /usr/local/bin/setpassword +COPY --from=build /out/migrate /usr/local/bin/migrate + +# The seed fixture, so `seed` works without a bind mount. +COPY seed/fixtures/seed.json /app/seed/fixtures/seed.json +ENV SEED_FIXTURE_PATH=/app/seed/fixtures/seed.json + +# The migration files, so an initContainer can apply them from this image. +# They are read-only at runtime and the process is non-root, so nothing here +# can be rewritten by the service. +COPY migrations/ /app/migrations/ +ENV MIGRATIONS_DIR=/app/migrations + +WORKDIR /app + +# Non-root. The service binds 8080, which needs no privilege, so there is no +# reason for this process to be able to write to its own filesystem or read +# another container's mounts. +USER krow:krow + +# 0.0.0.0, NOT the 127.0.0.1 that config.Load defaults to. A container that +# binds loopback is reachable only from inside itself — the symptom is a +# published port that refuses every connection while the process looks healthy. +ENV HTTP_HOST=0.0.0.0 \ + HTTP_PORT=8080 \ + APP_ENV=production \ + LOG_LEVEL=info + +EXPOSE 8080 + +# The check keys on the STATUS CODE, not on the body, because the two say +# different things on purpose (httpserver.handleHealth): +# +# 200 "ok" serving normally +# 200 "degraded" process healthy, schema unmigrated or dirty — still 200, +# because the fault is the database's and pulling this +# instance would not fix it +# 503 "unavailable" database unreachable +# +# Grepping the body for "ok" would mark a degraded instance unhealthy and take +# the whole deployment out of rotation during a migration window, which is +# exactly the behaviour that 200 was chosen to avoid. wget exits non-zero on +# 503 and zero on either 200, which is the intended semantics. +HEALTHCHECK --interval=15s --timeout=3s --start-period=20s --retries=3 \ + CMD wget -qO- http://127.0.0.1:8080/health >/dev/null 2>&1 || exit 1 + +# Exec form: the binary becomes PID 1 and receives SIGTERM directly, which is +# what cmd/api's signal.NotifyContext is waiting for to drain connections. +ENTRYPOINT ["/usr/local/bin/api"] diff --git a/infrastructure/docker-compose.local-db.yml b/infrastructure/docker-compose.local-db.yml new file mode 100644 index 0000000..5cd97fb --- /dev/null +++ b/infrastructure/docker-compose.local-db.yml @@ -0,0 +1,101 @@ +# ============================================================================ +# Overlay: run PostgreSQL alongside the API. +# +# docker compose -f docker-compose.yml -f docker-compose.local-db.yml up -d +# +# FOR STAGING AND FOR TESTING THE STACK ON ONE HOST. Not for production data. +# A database in a container on the same host as its application has no backup, +# no point-in-time recovery, no failover, and dies with the host. The base file +# assumes a managed PostgreSQL for exactly that reason. +# +# ── TLS ───────────────────────────────────────────────────────────────────── +# +# PostgreSQL is started with SSL on, using a self-signed certificate generated +# on first boot into the data volume. That is enough for DATABASE_SSLMODE= +# require, which encrypts the connection but does not verify who is on the +# other end — appropriate for a private compose network, NOT a substitute for +# verify-full against a real CA. +# +# It also means APP_ENV=production works here without weakening the check in +# internal/config, which is the point: the guard stays honest and the local +# stack meets it rather than dodging it. +# +# Set these in .env to match: +# DATABASE_HOST=postgres +# DATABASE_SSLMODE=require +# DATABASE_URL=postgres://krow:@postgres:5432/krow?sslmode=require +# ============================================================================ + +services: + postgres: + image: postgres:16-alpine + container_name: krow-postgres + environment: + POSTGRES_DB: ${DATABASE_NAME:-krow} + POSTGRES_USER: ${DATABASE_USER:-krow} + POSTGRES_PASSWORD: ${DATABASE_PASSWORD:?DATABASE_PASSWORD is required} + # scram-sha-256 rather than the image's md5 default. md5 has been + # deprecated upstream for years and is trivial to crack from a captured + # handshake. + POSTGRES_INITDB_ARGS: "--auth-host=scram-sha-256 --auth-local=scram-sha-256" + # Generate a self-signed certificate on first boot, then hand off to the + # image's own entrypoint with SSL enabled. The key must be 0600 and owned + # by the postgres user or the server refuses to start — that check is the + # reason this is done here rather than in a bind mount, where host + # ownership would leak in. + command: + - sh + - -c + - | + set -e + CERT=/var/lib/postgresql/data/server.crt + KEY=/var/lib/postgresql/data/server.key + if [ ! -f "$$CERT" ]; then + mkdir -p /var/lib/postgresql/data + openssl req -new -x509 -days 3650 -nodes -text \ + -out "$$CERT" -keyout "$$KEY" -subj "/CN=postgres" 2>/dev/null + chmod 0600 "$$KEY" + chown postgres:postgres "$$CERT" "$$KEY" + fi + exec docker-entrypoint.sh postgres \ + -c ssl=on -c ssl_cert_file="$$CERT" -c ssl_key_file="$$KEY" + volumes: + - postgres-data:/var/lib/postgresql/data + healthcheck: + # -U and -d so this reports on the application's database, not on + # whatever `postgres` happens to be reachable. + test: ["CMD-SHELL", "pg_isready -U ${DATABASE_USER:-krow} -d ${DATABASE_NAME:-krow}"] + interval: 10s + timeout: 5s + retries: 10 + start_period: 30s + ports: + # Loopback only. Publishing 5432 to the world is how a database ends up + # in someone else's botnet. + - "${POSTGRES_BIND:-127.0.0.1}:${POSTGRES_PORT:-5432}:5432" + restart: unless-stopped + stop_grace_period: 60s + security_opt: + - no-new-privileges:true + deploy: + resources: + limits: + memory: ${POSTGRES_MEMORY_LIMIT:-1G} + logging: + driver: json-file + options: + max-size: "10m" + max-file: "5" + networks: [krow] + + # The migration runner now has to wait for the database to accept + # connections, which it does not need to do when the database is managed and + # already up. + migrate: + depends_on: + postgres: + condition: service_healthy + +volumes: + postgres-data: + driver: local diff --git a/infrastructure/docker-compose.yml b/infrastructure/docker-compose.yml new file mode 100644 index 0000000..344a329 --- /dev/null +++ b/infrastructure/docker-compose.yml @@ -0,0 +1,142 @@ +# ============================================================================ +# Krow API — production stack +# +# cd infrastructure +# cp .env.docker.example .env +# docker compose up -d --build +# +# This file assumes a MANAGED PostgreSQL (RDS, Cloud SQL, Neon, a DBA's +# cluster) reachable from wherever these containers run. That is deliberate: +# a production database wants backups, point-in-time recovery, failover and a +# patch schedule, none of which a container in this file provides. +# +# To run PostgreSQL alongside the API — for staging, or to test the stack +# end-to-end on one host — layer the override on top: +# +# docker compose -f docker-compose.yml -f docker-compose.local-db.yml up -d +# +# ── ORDER OF OPERATIONS ───────────────────────────────────────────────────── +# +# `migrate` runs to completion BEFORE `api` starts, and `api` will not start if +# it fails. That is the expand/migrate/contract discipline from README.md +# expressed in the dependency graph rather than in a runbook: schema first, +# binary second, and every migration backwards-compatible with the version +# still running. +# +# ── TLS IS NOT OPTIONAL HERE ──────────────────────────────────────────────── +# +# internal/config rejects DATABASE_SSLMODE=disable when APP_ENV=production, so +# this stack will refuse to start against a database that cannot do TLS. That +# is the intended behaviour and not something to work around by dropping to +# APP_ENV=staging — a staging flag on a production deployment is a lie the next +# person has to discover. +# ============================================================================ + +name: krow + +# The database settings the API and the migration runner must agree on. +x-database-env: &database-env + DATABASE_HOST: ${DATABASE_HOST:?DATABASE_HOST is required} + DATABASE_PORT: ${DATABASE_PORT:-5432} + DATABASE_NAME: ${DATABASE_NAME:?DATABASE_NAME is required} + DATABASE_USER: ${DATABASE_USER:?DATABASE_USER is required} + DATABASE_PASSWORD: ${DATABASE_PASSWORD:?DATABASE_PASSWORD is required} + DATABASE_SCHEMA: ${DATABASE_SCHEMA:-public} + DATABASE_SSLMODE: ${DATABASE_SSLMODE:-require} + +x-logging: &logging + driver: json-file + options: + max-size: "10m" + max-file: "5" + +services: + # ── Schema ──────────────────────────────────────────────────────────────── + # One-shot. Exits 0 when the schema is current, including when there was + # nothing to apply. `api` waits on that exit code. + # + # DATABASE_URL is a single pre-built string rather than assembled from the + # parts above because golang-migrate takes a URL, and a password containing + # @ : / ? or # must be percent-encoded in one. Compose cannot encode it, so + # the encoding is the operator's job — see .env.docker.example. + migrate: + image: migrate/migrate:v4.19.0 + container_name: krow-migrate + volumes: + - ../migrations:/migrations:ro + command: + - -path=/migrations + - -database=${DATABASE_URL:?DATABASE_URL is required — see .env.docker.example} + - up + restart: "no" + logging: *logging + networks: [krow] + + # ── API ─────────────────────────────────────────────────────────────────── + api: + build: + context: .. + dockerfile: infrastructure/Dockerfile.api + args: + VERSION: ${VERSION:-dev} + image: ${IMAGE:-doormile/krowbackend:latest} + container_name: krow-api + depends_on: + migrate: + condition: service_completed_successfully + environment: + <<: *database-env + APP_ENV: ${APP_ENV:-production} + LOG_LEVEL: ${LOG_LEVEL:-info} + # 0.0.0.0 so the published port actually reaches the process. The binary + # defaults to 127.0.0.1, which inside a container means "nobody". + HTTP_HOST: 0.0.0.0 + HTTP_PORT: "8080" + HTTP_READ_TIMEOUT: ${HTTP_READ_TIMEOUT:-15s} + HTTP_WRITE_TIMEOUT: ${HTTP_WRITE_TIMEOUT:-30s} + HTTP_IDLE_TIMEOUT: ${HTTP_IDLE_TIMEOUT:-60s} + HTTP_SHUTDOWN_TIMEOUT: ${HTTP_SHUTDOWN_TIMEOUT:-10s} + # Browser origins allowed to call this API. Empty means same-origin only, + # and the CORS middleware is then not installed at all. Cookies are the + # auth mechanism, so any origin listed here can hold a session. + # + # "*" is refused at startup: browsers reject Allow-Origin "*" together + # with credentials, so it would break authenticated calls rather than + # loosen anything. + HTTP_CORS_ORIGINS: ${HTTP_CORS_ORIGINS:-} + # lax | none | strict. "none" is required when the frontend is on a + # different registrable domain from the API — otherwise the browser + # withholds the cookie however correct the CORS headers are. + HTTP_COOKIE_SAMESITE: ${HTTP_COOKIE_SAMESITE:-lax} + DATABASE_MAX_OPEN_CONNS: ${DATABASE_MAX_OPEN_CONNS:-25} + DATABASE_MIN_IDLE_CONNS: ${DATABASE_MIN_IDLE_CONNS:-2} + DATABASE_CONN_MAX_LIFETIME: ${DATABASE_CONN_MAX_LIFETIME:-30m} + DATABASE_CONNECT_TIMEOUT: ${DATABASE_CONNECT_TIMEOUT:-10s} + DATABASE_STATEMENT_TIMEOUT: ${DATABASE_STATEMENT_TIMEOUT:-10s} + ports: + # Bound to loopback by default. A reverse proxy in front of this is what + # terminates TLS; publishing 0.0.0.0:8080 would put a plain-HTTP service + # carrying session cookies straight onto the network. + - "${API_BIND:-127.0.0.1}:${API_PORT:-8080}:8080" + restart: unless-stopped + stop_grace_period: 30s + read_only: true + security_opt: + - no-new-privileges:true + cap_drop: + - ALL + tmpfs: + - /tmp:size=32m + deploy: + resources: + limits: + cpus: "${API_CPU_LIMIT:-2}" + memory: ${API_MEMORY_LIMIT:-512M} + reservations: + memory: 128M + logging: *logging + networks: [krow] + +networks: + krow: + driver: bridge diff --git a/scripts/drop_public_tables.go b/scripts/drop_public_tables.go new file mode 100644 index 0000000..5f4e5aa --- /dev/null +++ b/scripts/drop_public_tables.go @@ -0,0 +1,147 @@ +//go:build ignore + +// Command drop_public_tables removes every table in krowdb's `public` schema. +// +// DESTRUCTIVE AND IRREVERSIBLE. Authorised by the database admin on 2026-08-24 +// to clear an unrelated delivery-platform schema (132 tables, 222 MB) so the +// Krow backend can take over this database. +// +// `hdb_catalog` — Hasura's own 8 metadata tables — is deliberately NOT touched. +// +// A structure-only snapshot of what this removes was taken beforehand: +// /Users/tenext/Documents/Krow/krowdb_public_snapshot_2026-08-24.sql +// It carries DDL, not rows. Row data is NOT recoverable after this runs. +// +// Everything happens in ONE transaction: it either clears the schema completely +// or leaves it exactly as it was. +// +// Run: +// cd go-api && go run ../scripts/drop_public_tables.go +// +// It reads DATABASE_* from the environment (the Makefile exports .env), and +// requires CONFIRM_DROP=yes so it cannot fire by accident. +package main + +import ( + "context" + "fmt" + "net/url" + "os" + "time" + + "github.com/jackc/pgx/v5" +) + +func env(k, def string) string { + if v := os.Getenv(k); v != "" { + return v + } + return def +} + +func main() { + if os.Getenv("CONFIRM_DROP") != "yes" { + fmt.Println("refusing: set CONFIRM_DROP=yes to run this.") + fmt.Println("this drops EVERY table in the target database's public schema.") + os.Exit(1) + } + + dsn := fmt.Sprintf("postgres://%s:%s@%s:%s/%s?sslmode=%s&connect_timeout=15", + url.QueryEscape(env("DATABASE_USER", "admin")), + url.QueryEscape(env("DATABASE_PASSWORD", "")), + env("DATABASE_HOST", "127.0.0.1"), + env("DATABASE_PORT", "5432"), + url.QueryEscape(env("DATABASE_NAME", "krowdb")), + env("DATABASE_SSLMODE", "disable"), + ) + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute) + defer cancel() + + conn, err := pgx.Connect(ctx, dsn) + if err != nil { + fmt.Println("connect failed:", err) + os.Exit(1) + } + defer conn.Close(ctx) + + var dbname string + _ = conn.QueryRow(ctx, `SELECT current_database()`).Scan(&dbname) + fmt.Println("connected to:", dbname) + + tx, err := conn.Begin(ctx) + if err != nil { + fmt.Println("begin failed:", err) + os.Exit(1) + } + defer tx.Rollback(ctx) + + names, err := collect(ctx, tx, ` + SELECT c.relname FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'r' ORDER BY 1`) + if err != nil { + fmt.Println("listing tables failed:", err) + os.Exit(1) + } + views, _ := collect(ctx, tx, ` + SELECT c.relname FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace + WHERE n.nspname = 'public' AND c.relkind = 'v'`) + seqs, _ := collect(ctx, tx, ` + SELECT sequence_name FROM information_schema.sequences WHERE sequence_schema = 'public'`) + + fmt.Printf("dropping %d tables, %d views, %d sequences from public\n", + len(names), len(views), len(seqs)) + fmt.Println("hdb_catalog (Hasura) is NOT touched.") + + // CASCADE because the schema carries foreign keys between these tables; + // dropping in dependency order would otherwise be required. + for _, t := range names { + if _, err := tx.Exec(ctx, fmt.Sprintf(`DROP TABLE IF EXISTS public.%q CASCADE`, t)); err != nil { + fmt.Printf(" failed on %s: %v\n", t, err) + fmt.Println("rolling back — nothing was dropped.") + os.Exit(1) + } + } + for _, v := range views { + if _, err := tx.Exec(ctx, fmt.Sprintf(`DROP VIEW IF EXISTS public.%q CASCADE`, v)); err != nil { + fmt.Printf(" failed on view %s: %v\n", v, err) + os.Exit(1) + } + } + for _, s := range seqs { + if _, err := tx.Exec(ctx, fmt.Sprintf(`DROP SEQUENCE IF EXISTS public.%q CASCADE`, s)); err != nil { + fmt.Printf(" failed on sequence %s: %v\n", s, err) + os.Exit(1) + } + } + + if err := tx.Commit(ctx); err != nil { + fmt.Println("commit failed:", err) + os.Exit(1) + } + fmt.Println("committed.") + + var pub, hdb int + _ = conn.QueryRow(ctx, `SELECT count(*) FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace WHERE n.nspname='public' AND c.relkind='r'`).Scan(&pub) + _ = conn.QueryRow(ctx, `SELECT count(*) FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace WHERE n.nspname='hdb_catalog' AND c.relkind='r'`).Scan(&hdb) + fmt.Println("\nafter:") + fmt.Println(" public tables :", pub) + fmt.Println(" hdb_catalog tables :", hdb, "(preserved)") +} + +func collect(ctx context.Context, tx pgx.Tx, q string) ([]string, error) { + rows, err := tx.Query(ctx, q) + if err != nil { + return nil, err + } + defer rows.Close() + var out []string + for rows.Next() { + var s string + if err := rows.Scan(&s); err != nil { + return nil, err + } + out = append(out, s) + } + return out, rows.Err() +}