1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 | #!/usr/bin/env bash
#
# Fast auto-deploy for the standalone box. Installed as a ~60s systemd timer by
# scripts/standalone-deploy.sh. Polls the deploy branch; when new commits land,
# it pulls, rebuilds, and runs migrations. Push -> live in ~1-2 min, hands-off.
#
# Exits immediately (cheap) when there is nothing new, so a tight interval is fine.
#
# Rollback (2026-07-21): a bad deploy used to leave the box with the OLD
# container already destroyed and the NEW one unhealthy -- a failed health
# check just exited 1 with nothing serving traffic until a human noticed and
# fixed it by hand. The previous image is now tagged before every build, and
# a failed health check re-deploys it automatically instead of leaving a
# broken container live. A SHA that fails is remembered (FAILED_MARKER) so a
# persistently broken commit doesn't retry-and-fail every 60s forever -- it
# waits for a newer commit (or a human clearing the marker) instead.
set -euo pipefail
REPO_DIR="/opt/gluecron"
BRANCH="main"
COMPOSE="docker compose -f docker-compose.standalone.yml"
IMAGE="gluecron-gluecron"
FAILED_MARKER="$REPO_DIR/.last-failed-deploy-sha"
cd "$REPO_DIR"
git fetch origin "$BRANCH" --quiet
local_sha=$(git rev-parse HEAD)
remote_sha=$(git rev-parse "origin/$BRANCH")
[ "$local_sha" = "$remote_sha" ] && exit 0
if [ -f "$FAILED_MARKER" ] && [ "$(cat "$FAILED_MARKER")" = "$remote_sha" ]; then
# Already tried and rolled back from this exact commit -- don't loop.
exit 0
fi
echo "$(date -Is) deploying $local_sha -> $remote_sha"
# Preserve the last-known-good image BEFORE touching anything, so a bad
# deploy can roll back to exactly what was running a moment ago. `|| true`:
# the very first deploy on a fresh box has no prior image to tag.
docker tag "$IMAGE:latest" "$IMAGE:last-good" 2>/dev/null || true
prev_sha="$local_sha"
git reset --hard "origin/$BRANCH" # untracked .env / backups are preserved
# Build provenance for the container. The image ships without a .git dir, so
# src/lib/build-info.ts can only report a real SHA if we hand it one via env.
# Without this the footer renders "unknown · unknown", /api/version reports a
# SHA that never changes (so the client auto-update banner can never fire),
# and the PWA service-worker cache key never rotates between deploys.
# docker-compose.standalone.yml interpolates these at `up` time.
GIT_SHA="$(git rev-parse HEAD)"
GIT_BRANCH="$BRANCH"
BUILD_SHA="$GIT_SHA"
export GIT_SHA GIT_BRANCH BUILD_SHA
# NOTE: `|| true` is deliberate. On this Coolify co-tenant box the compose
# also defines a `caddy` service that tries to bind host :80/:443, which
# Coolify's proxy already owns — so `up` reports a non-zero exit for caddy
# even though the app (gluecron) started fine. Under `set -e` that would abort
# the deploy BEFORE the coolify reattach + health gate below. We tolerate the
# partial failure here and instead gate on the app's OWN health further down.
$COMPOSE up -d --build || echo "$(date -Is) compose up returned non-zero (likely the co-tenant caddy port conflict) — continuing to app health gate"
# Co-tenant ingress reattach (Coolify boxes only).
# `docker compose up --build` recreates the gluecron container, which DROPS
# any network attachment not declared in docker-compose.standalone.yml. On
# this box the public ingress is Coolify's Traefik ("coolify-proxy"), which
# reaches the app over the external "coolify" network via the file route
# /traefik/dynamic/gluecron.yaml in the coolify-proxy container. Without this
# reattach, every deploy 502s the site until someone reconnects by hand.
# Idempotent: a no-op if already attached; skipped entirely on a dedicated VPS
# that has no "coolify" network. Traefik's file provider watches with fsnotify
# (--providers.file.watch=true), so it notices the container is reachable
# again within a second or two of the reconnect — no manual nudge needed.
if docker network inspect coolify >/dev/null 2>&1; then
docker network connect coolify gluecron-gluecron-1 2>/dev/null \
&& echo "$(date -Is) reattached gluecron to coolify network" \
|| echo "$(date -Is) coolify network already attached (ok)"
fi
$COMPOSE exec -T gluecron bun run db:migrate || true
docker image prune -f >/dev/null 2>&1 || true
# App health gate — the real success signal (not caddy). Poll the container's
# own /healthz; if the app itself never comes up, roll back instead of
# leaving a broken deploy live.
healthy=0
for _ in $(seq 1 20); do
if docker exec gluecron-gluecron-1 wget -qO- --timeout=4 http://localhost:3000/healthz >/dev/null 2>&1; then
healthy=1; break
fi
sleep 3
done
if [ "$healthy" = "1" ]; then
rm -f "$FAILED_MARKER"
echo "$(date -Is) deploy complete: $remote_sha (app healthy)"
exit 0
fi
echo "$(date -Is) DEPLOY FAILED: app /healthz never came up for $remote_sha" >&2
docker logs --tail 80 gluecron-gluecron-1 >&2 || true
if ! docker image inspect "$IMAGE:last-good" >/dev/null 2>&1; then
echo "$(date -Is) no last-good image to roll back to (first deploy on this box?) — human intervention required" >&2
echo "$remote_sha" > "$FAILED_MARKER"
exit 1
fi
echo "$(date -Is) rolling back to $prev_sha" >&2
git reset --hard "$prev_sha"
docker tag "$IMAGE:last-good" "$IMAGE:latest"
$COMPOSE up -d --no-build || echo "$(date -Is) compose up (rollback) returned non-zero (likely the co-tenant caddy port conflict) — continuing to health gate"
if docker network inspect coolify >/dev/null 2>&1; then
docker network connect coolify gluecron-gluecron-1 2>/dev/null || true
fi
rb_healthy=0
for _ in $(seq 1 20); do
if docker exec gluecron-gluecron-1 wget -qO- --timeout=4 http://localhost:3000/healthz >/dev/null 2>&1; then
rb_healthy=1; break
fi
sleep 3
done
echo "$remote_sha" > "$FAILED_MARKER"
if [ "$rb_healthy" = "1" ]; then
echo "$(date -Is) rollback to $prev_sha succeeded — app healthy again. $remote_sha will not be retried automatically; push a fix or clear $FAILED_MARKER to try again." >&2
exit 1
fi
echo "$(date -Is) ROLLBACK ALSO FAILED — human intervention required NOW (site may be down)" >&2
exit 1
|