#!/usr/bin/env bash # Deploy the server to the production host, from this checkout. # # Until this existed the deployment was manual to a box nobody had written # down, and production ran code months behind main - three of the desktop # app's screens talked to routes that were not there. A deploy that is a # script gets run; one that is a memory does not. # # server/deploy.sh build, back up the database, migrate, switch # BASELINE=3 server/deploy.sh first run against a database that predates # migration tracking: adopt 1..3 unrun # DRY_RUN=1 server/deploy.sh build and ship, touch nothing running # # What it does, in order, and why the order matters: # 1. builds the head-office web app INTO the Go module, then a static # linux/amd64 binary here - the host has 3.6 GB of RAM and must not compile # 2. pg_dumps the database to backups/ on the host BEFORE anything changes # 3. builds the runtime-only image on the host from the shipped binary # 4. runs `migrate` with the NEW binary while the OLD server still serves; # a failing migration therefore stops here with production untouched # 5. switches the container, then proves the routes answer over the public URL set -euo pipefail cd "$(dirname "$0")" HOST=${HOST:-root@66.116.226.161} KEY=${KEY:-$HOME/.ssh/behavision_deploy} REMOTE_DIR=/root/behavision PUBLIC=https://mcp.loyaly.ai SSH=(ssh -i "$KEY" -o BatchMode=yes -o ConnectTimeout=10 "$HOST") # Go is not always on an interactive shell's PATH - a Homebrew or tarball # install lands in a directory that .zprofile adds but a script does not # inherit, so this failed at step 1 with "go: command not found" on the very # machine it was written on. Found the only way it could be: by somebody # running it. A deploy that needs the operator to fix their environment first # is a deploy that gets skipped. for d in "$HOME/go/bin" /usr/local/go/bin /opt/homebrew/bin; do [ -x "$d/go" ] && case ":$PATH:" in *":$d:"*) ;; *) PATH="$PATH:$d";; esac done command -v go >/dev/null || { echo "go not found - install it or add it to PATH" >&2; exit 1; } VERSION=$(git describe --tags --always --dirty) case "$VERSION" in *-dirty) echo "refusing to deploy uncommitted changes ($VERSION)" >&2; exit 1;; esac step() { printf '\n\033[1m%s\033[0m\n' "$*"; } step "1. Build $VERSION" (cd ../web && npm run build >/dev/null) CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath \ -ldflags "-s -w -X main.version=${VERSION}" -o /tmp/behavision-server ./cmd/behavision-server ls -la /tmp/behavision-server | awk '{print " " $5 " bytes"}' step "2. Ship" "${SSH[@]}" "mkdir -p $REMOTE_DIR/release/$VERSION $REMOTE_DIR/backups" scp -q -i "$KEY" /tmp/behavision-server Dockerfile.runtime "$HOST:$REMOTE_DIR/release/$VERSION/" git rev-parse HEAD | "${SSH[@]}" "cat > $REMOTE_DIR/release/$VERSION/GIT_SHA" if [ "${DRY_RUN:-}" != "" ]; then echo "DRY_RUN: shipped to $REMOTE_DIR/release/$VERSION, nothing changed"; exit 0; fi step "3. Back up the database" # The dump's own filename is echoed, not `ls | tail -1`, which reported the # WRONG file: ls sorts alphabetically, so pre-...-demo-12-... sorts before # pre-...-demo-6-... and the line printed a backup from four days earlier. A # deploy that names the wrong safety net is worse than one that names none - # that is the file somebody reaches for at the worst possible moment. # # Bare `-s` on the dump so an empty or failed one cannot be reported as a # backup: pg_dump exiting non-zero already fails the pipeline under pipefail, # but a zero-byte gzip would still satisfy it. "${SSH[@]}" "set -e; f=$REMOTE_DIR/backups/pre-$VERSION-\$(date +%Y%m%d-%H%M%S).sql.gz; \ docker exec behavision-db sh -c 'PGPASSWORD=\$POSTGRES_PASSWORD pg_dump -U behavision -d behavision' | gzip > \$f; \ [ -s \$f ] || { echo 'backup is empty - refusing to continue' >&2; exit 1; }; \ ls -la \$f" step "4. Image" "${SSH[@]}" "cd $REMOTE_DIR/release/$VERSION && docker build -q -t behavision-backend:$VERSION -f Dockerfile.runtime . && docker tag behavision-backend:$VERSION behavision-backend:latest" step "5. Migrate (old server still serving)" # `run` uses the compose service's environment and network, so the new binary # reaches postgres exactly as the server will. --no-deps: do not restart the # broker or the database to run a migration. if [ -n "${BASELINE:-}" ]; then "${SSH[@]}" "cd $REMOTE_DIR && docker compose run --rm --no-deps -T backend migrate -baseline $BASELINE" fi "${SSH[@]}" "cd $REMOTE_DIR && docker compose run --rm --no-deps -T backend migrate && docker compose run --rm --no-deps -T backend migrate -status" step "6. Switch" "${SSH[@]}" "cd $REMOTE_DIR && docker compose up -d --no-build --no-deps backend && sleep 4 && docker logs --tail 15 behavision-backend" step "7. Verify over $PUBLIC" # 401 is a PASS, and that distinction is the whole point of this step. An # unauthenticated call to a route that EXISTS is refused; a route the binary # never registered is a 404. So this proves the routing rather than the auth - # which is precisely what a deploy gets wrong, and what otherwise surfaces as a # console showing "Backend integration required" against an API that shipped. # # The uuid matches nothing on purpose: the admin drill-down must answer 401 # with no session, never 404. NOBODY=00000000-0000-4000-8000-000000000000 fail=0 for p in /healthz \ /api/admin/clients \ "/api/admin/clients/$NOBODY" \ "/api/admin/clients/$NOBODY/sites" \ "/api/admin/clients/$NOBODY/sites/x/cameras" \ /api/admin/monitoring/summary \ /api/sales \ /api/sales/x \ /api/dashboard/summary \ /api/team /api/visits /api/cameras; do code=$(curl -s -o /dev/null -w '%{http_code}' -m 15 "$PUBLIC$p") case "$code" in 200|401) verdict="ok" ;; 404) verdict="MISSING - this binary does not serve that route"; fail=1 ;; *) verdict="unexpected"; fail=1 ;; esac printf ' %-46s %s %s\n' "$p" "$code" "$verdict" done curl -s -m 15 "$PUBLIC/healthz" | head -c 300; echo [ "$fail" = 0 ] || { echo; echo "VERIFY FAILED - routes above marked MISSING did not ship" >&2; exit 1; }