384 lines
13 KiB
Go
384 lines
13 KiB
Go
// Package httpserver holds the HTTP surface.
|
|
//
|
|
// It serves /health, the sign-in endpoints, the entity endpoints described in
|
|
// docs/api-contract.md, and the current-user endpoints.
|
|
//
|
|
// Phase 3C replaced the development identity with real authentication. Every
|
|
// request outside the small public allowlist in auth.go must carry a session
|
|
// cookie; the middleware resolves it to a user row and puts that user, and
|
|
// their organization, on the request context. Nothing downstream changed —
|
|
// every service and repository already took the organization as a parameter,
|
|
// which is what devOrgMiddleware existed to make true.
|
|
//
|
|
// Authorization is NOT here. A signed-in user reaches every endpoint they could
|
|
// reach before; deciding which roles may do what is Phase 3D.
|
|
package httpserver
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"log/slog"
|
|
"net"
|
|
"net/http"
|
|
"strconv"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/auth"
|
|
"github.com/krow/krow-backend/go-api/internal/config"
|
|
"github.com/krow/krow-backend/go-api/internal/db"
|
|
"github.com/krow/krow-backend/go-api/internal/service"
|
|
)
|
|
|
|
// Server binds the router, the pool, authentication and the lifecycle together.
|
|
type Server struct {
|
|
cfg *config.Config
|
|
db *db.DB
|
|
api *service.Registry
|
|
definitions *service.DefinitionsService
|
|
log *slog.Logger
|
|
http *http.Server
|
|
started time.Time
|
|
endpoints int
|
|
|
|
// The authentication surface. sessions owns the lifecycle, users is the
|
|
// read side of the users table, credentials verifies a password against it,
|
|
// and the two limiters bound how often that may be attempted.
|
|
//
|
|
// Two limiters, not one, because the budgets are different sizes on
|
|
// purpose: an email is one account and gets a tight budget, while an
|
|
// address may be a whole office behind NAT and gets a loose one. Sharing a
|
|
// limiter would force the office to live within one person's budget.
|
|
sessions *auth.Manager
|
|
users auth.UserStore
|
|
credentials *auth.Credentials
|
|
loginByEmail *attemptLimiter
|
|
loginByAddr *attemptLimiter
|
|
|
|
// now is injectable so tests can drive expiry without sleeping.
|
|
now func() time.Time
|
|
}
|
|
|
|
// Option adjusts the server before it is wired. Production passes none.
|
|
type Option func(*serverOptions)
|
|
|
|
type serverOptions struct {
|
|
policy auth.Policy
|
|
now func() time.Time
|
|
perEmail int
|
|
perAddress int
|
|
loginWindow time.Duration
|
|
}
|
|
|
|
// WithSessionPolicy overrides the session lifetimes. For tests that need to
|
|
// reach an expiry without waiting twelve hours for it.
|
|
func WithSessionPolicy(p auth.Policy) Option {
|
|
return func(o *serverOptions) { o.policy = p }
|
|
}
|
|
|
|
// WithClock replaces the clock used for session expiry and last_login_at.
|
|
func WithClock(now func() time.Time) Option {
|
|
return func(o *serverOptions) {
|
|
if now != nil {
|
|
o.now = now
|
|
}
|
|
}
|
|
}
|
|
|
|
// WithLoginRateLimit overrides the failed-attempt budgets and their window.
|
|
//
|
|
// perEmail bounds attempts against one account; perAddress bounds attempts from
|
|
// one client address across all accounts. Both are consulted on every attempt.
|
|
func WithLoginRateLimit(perEmail, perAddress int, window time.Duration) Option {
|
|
return func(o *serverOptions) {
|
|
o.perEmail, o.perAddress, o.loginWindow = perEmail, perAddress, window
|
|
}
|
|
}
|
|
|
|
// New wires the routes and returns a server that has not yet been started.
|
|
//
|
|
// Authentication is built here rather than passed in, so there is exactly one
|
|
// construction of the session manager and no way to start a server with the
|
|
// middleware wired to a different store than the login handler.
|
|
func New(cfg *config.Config, database *db.DB, log *slog.Logger, opts ...Option) (*Server, error) {
|
|
o := serverOptions{
|
|
policy: auth.DefaultPolicy,
|
|
now: time.Now,
|
|
perEmail: loginAttemptLimit,
|
|
perAddress: loginAddressLimit,
|
|
loginWindow: loginAttemptWindow,
|
|
}
|
|
for _, opt := range opts {
|
|
opt(&o)
|
|
}
|
|
|
|
sessions, err := auth.NewManager(auth.NewPGStore(database.Pool), o.policy)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("build session manager: %w", err)
|
|
}
|
|
sessions.WithClock(o.now)
|
|
|
|
users := auth.NewPGUserStore(database.Pool)
|
|
s := &Server{
|
|
cfg: cfg, db: database, log: log,
|
|
api: service.NewRegistry(database.Pool),
|
|
definitions: service.NewDefinitions(database.Pool),
|
|
started: o.now(),
|
|
sessions: sessions,
|
|
users: users,
|
|
credentials: auth.NewCredentials(users),
|
|
loginByEmail: newAttemptLimiter(o.perEmail, o.loginWindow, o.now),
|
|
loginByAddr: newAttemptLimiter(o.perAddress, o.loginWindow, o.now),
|
|
now: o.now,
|
|
}
|
|
|
|
mux := http.NewServeMux()
|
|
mux.HandleFunc("GET /health", s.handleHealth)
|
|
s.endpoints = s.routeAuth(mux) + s.routeResources(mux) + s.routeMe(mux) + s.routeDefinitions(mux)
|
|
|
|
handler := jsonErrors(mux)
|
|
// Authentication sits where devOrgMiddleware used to, so every route below
|
|
// it — including the mux's own 404 — is behind the allowlist.
|
|
handler = s.authenticate(handler)
|
|
handler = recoverer(log)(handler)
|
|
// CORS sits outside the recoverer so a preflight is answered without
|
|
// touching the router, and inside the logger so refused origins are still
|
|
// visible in the log. With no allowlist configured it is not installed at
|
|
// all, which is the same-origin default.
|
|
if len(cfg.HTTP.CORSOrigins) > 0 {
|
|
handler = cors(cfg.HTTP.CORSOrigins)(handler)
|
|
}
|
|
handler = requestLogger(log)(handler)
|
|
|
|
s.http = &http.Server{
|
|
Addr: net.JoinHostPort(cfg.HTTP.Host, strconv.Itoa(cfg.HTTP.Port)),
|
|
Handler: handler,
|
|
ReadTimeout: cfg.HTTP.ReadTimeout,
|
|
WriteTimeout: cfg.HTTP.WriteTimeout,
|
|
IdleTimeout: cfg.HTTP.IdleTimeout,
|
|
}
|
|
return s, nil
|
|
}
|
|
|
|
// Sessions exposes the session manager, so the process can sweep expired rows
|
|
// and tests can drive the clock.
|
|
func (s *Server) Sessions() *auth.Manager { return s.sessions }
|
|
|
|
// Handler exposes the routed handler so tests can drive it without a listener.
|
|
func (s *Server) Handler() http.Handler { return s.http.Handler }
|
|
|
|
// Endpoints is how many routes were registered.
|
|
func (s *Server) Endpoints() int { return s.endpoints }
|
|
|
|
// Addr is the address the server listens on.
|
|
func (s *Server) Addr() string { return s.http.Addr }
|
|
|
|
// Start blocks until the server stops accepting connections.
|
|
func (s *Server) Start() error {
|
|
err := s.http.ListenAndServe()
|
|
if errors.Is(err, http.ErrServerClosed) {
|
|
return nil
|
|
}
|
|
return err
|
|
}
|
|
|
|
// Shutdown drains in-flight requests, then gives up after the configured grace.
|
|
func (s *Server) Shutdown(ctx context.Context) error {
|
|
ctx, cancel := context.WithTimeout(ctx, s.cfg.HTTP.ShutdownTimeout)
|
|
defer cancel()
|
|
return s.http.Shutdown(ctx)
|
|
}
|
|
|
|
// healthResponse is the entire public /health body: one field, deliberately.
|
|
//
|
|
// /health is unauthenticated and reachable by anyone who can reach the port,
|
|
// so it is treated as a public document rather than as an operator's console.
|
|
// Everything an unauthenticated caller legitimately needs is the answer to
|
|
// "should traffic be sent here", and that fits in a status string plus the
|
|
// HTTP status code.
|
|
//
|
|
// What used to be here and is now deliberately absent: the PostgreSQL version,
|
|
// the database name, the schema name, the applied migration version, the table
|
|
// count, the connection error text, the deployment environment and the process
|
|
// uptime. Individually each is small; together they are a free reconnaissance
|
|
// report — the server version to look up known CVEs against, the migration
|
|
// version to date the deployment, the table count and error text to infer
|
|
// shape and topology. None of it is diagnostic to anyone who could not already
|
|
// read it from the database directly.
|
|
//
|
|
// The check itself is unchanged. db.Check still runs on every request and
|
|
// still decides the answer; its full detail now goes to the server log, where
|
|
// the operator is, instead of into the response, where the internet is. See
|
|
// logHealth.
|
|
type healthResponse struct {
|
|
Status string `json:"status"`
|
|
}
|
|
|
|
// handleHealth reports whether this instance should be sent traffic.
|
|
//
|
|
// 200 "ok" serving normally
|
|
// 200 "degraded" the process is healthy, the schema is not: unmigrated,
|
|
// or a migration left the version dirty. Still 200,
|
|
// because the fault is the database's and taking the
|
|
// instance out of rotation would not fix it.
|
|
// 503 "unavailable" the database is unreachable, so a load balancer can act
|
|
// on the status code alone without parsing the body.
|
|
//
|
|
// The three status words are a coarse operational signal, not infrastructure
|
|
// detail: they say what a caller should do, and nothing about what is running.
|
|
func (s *Server) handleHealth(w http.ResponseWriter, r *http.Request) {
|
|
ctx, cancel := context.WithTimeout(r.Context(), 5*time.Second)
|
|
defer cancel()
|
|
|
|
health := s.db.Check(ctx)
|
|
|
|
status, code := "ok", http.StatusOK
|
|
switch {
|
|
case !health.Reachable:
|
|
status, code = "unavailable", http.StatusServiceUnavailable
|
|
case health.MigrationDirty, !health.SchemaPresent:
|
|
status = "degraded"
|
|
}
|
|
|
|
s.logHealth(status, health)
|
|
|
|
w.Header().Set("Content-Type", "application/json; charset=utf-8")
|
|
w.Header().Set("Cache-Control", "no-store")
|
|
w.WriteHeader(code)
|
|
enc := json.NewEncoder(w)
|
|
enc.SetIndent("", " ")
|
|
_ = enc.Encode(healthResponse{Status: status})
|
|
}
|
|
|
|
// logHealth writes the detail the response body used to carry.
|
|
//
|
|
// This is the "internally" half of the change: nothing was deleted from
|
|
// db.Check, and nothing it learns is thrown away — the audience moved from the
|
|
// response to the log, which is already authenticated by virtue of being on
|
|
// the host.
|
|
//
|
|
// A load balancer polls this endpoint every few seconds, so a healthy check
|
|
// logs at debug and a bad one at warn. Anything other than "ok" is worth
|
|
// seeing without turning debug on.
|
|
func (s *Server) logHealth(status string, h db.Health) {
|
|
attrs := []any{
|
|
"status", status,
|
|
"env", s.cfg.AppEnv,
|
|
"uptime_seconds", int64(time.Since(s.started).Seconds()),
|
|
"reachable", h.Reachable,
|
|
"schema", h.Schema,
|
|
"schema_present", h.SchemaPresent,
|
|
"table_count", h.TableCount,
|
|
"migration_dirty", h.MigrationDirty,
|
|
"latency_ms", h.LatencyMS,
|
|
}
|
|
if h.Database != "" {
|
|
attrs = append(attrs, "database", h.Database, "postgres_version", h.Version)
|
|
}
|
|
if h.AppliedMigration != nil {
|
|
attrs = append(attrs, "applied_migration", *h.AppliedMigration)
|
|
}
|
|
if h.Error != "" {
|
|
attrs = append(attrs, "error", h.Error)
|
|
}
|
|
|
|
if status == "ok" {
|
|
s.log.Debug("health", attrs...)
|
|
return
|
|
}
|
|
s.log.Warn("health", attrs...)
|
|
}
|
|
|
|
type statusRecorder struct {
|
|
http.ResponseWriter
|
|
code int
|
|
}
|
|
|
|
func (r *statusRecorder) WriteHeader(code int) {
|
|
r.code = code
|
|
r.ResponseWriter.WriteHeader(code)
|
|
}
|
|
|
|
func requestLogger(log *slog.Logger) func(http.Handler) http.Handler {
|
|
return func(next http.Handler) http.Handler {
|
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
started := time.Now()
|
|
rec := &statusRecorder{ResponseWriter: w, code: http.StatusOK}
|
|
next.ServeHTTP(rec, r)
|
|
log.Info("request",
|
|
"method", r.Method, "path", r.URL.Path,
|
|
"status", rec.code, "duration_ms", time.Since(started).Milliseconds())
|
|
})
|
|
}
|
|
}
|
|
|
|
// jsonErrors converts net/http's own plain-text 404 and 405 replies into the
|
|
// documented error envelope.
|
|
//
|
|
// ServeMux writes those itself, before any handler of ours runs, so a client
|
|
// that hit a wrong path or method would otherwise get "404 page not found" in
|
|
// text/plain while every other response is JSON. Only the mux's own replies are
|
|
// rewritten: anything that set a content type has already answered properly.
|
|
func jsonErrors(next http.Handler) http.Handler {
|
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
iw := &interceptor{ResponseWriter: w}
|
|
next.ServeHTTP(iw, r)
|
|
if !iw.rewritten || iw.wrote {
|
|
return
|
|
}
|
|
errCode, message := "not_found", "resource not found"
|
|
if iw.code == http.StatusMethodNotAllowed {
|
|
errCode = "method_not_allowed"
|
|
message = r.Method + " is not supported for this resource"
|
|
}
|
|
writeJSON(w, iw.code, errorEnvelope{Error: errorBody{
|
|
Code: errCode, Message: message, Details: map[string]string{},
|
|
}})
|
|
})
|
|
}
|
|
|
|
// interceptor defers the mux's plain-text 404/405 body so it can be replaced.
|
|
type interceptor struct {
|
|
http.ResponseWriter
|
|
code int
|
|
rewritten bool // this is a mux-generated 404/405 we intend to replace
|
|
wrote bool // a body already went to the client
|
|
}
|
|
|
|
func (i *interceptor) WriteHeader(code int) {
|
|
i.code = code
|
|
if code == http.StatusNotFound || code == http.StatusMethodNotAllowed {
|
|
if i.Header().Get("Content-Type") != "application/json; charset=utf-8" {
|
|
i.rewritten = true
|
|
return // hold the header back; jsonErrors writes its own
|
|
}
|
|
}
|
|
i.ResponseWriter.WriteHeader(code)
|
|
}
|
|
|
|
func (i *interceptor) Write(b []byte) (int, error) {
|
|
if i.rewritten {
|
|
return len(b), nil // swallow the mux's plain-text body
|
|
}
|
|
i.wrote = true
|
|
return i.ResponseWriter.Write(b)
|
|
}
|
|
|
|
// recoverer turns a panic into a logged 500 rather than a dropped connection.
|
|
func recoverer(log *slog.Logger) func(http.Handler) http.Handler {
|
|
return func(next http.Handler) http.Handler {
|
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
defer func() {
|
|
if v := recover(); v != nil {
|
|
log.Error("panic", "value", v, "path", r.URL.Path)
|
|
writeJSON(w, http.StatusInternalServerError, errorEnvelope{Error: errorBody{
|
|
Code: "internal", Message: "internal error", Details: map[string]string{},
|
|
}})
|
|
}
|
|
}()
|
|
next.ServeHTTP(w, r)
|
|
})
|
|
}
|
|
}
|