127 lines
5.7 KiB
Bash
127 lines
5.7 KiB
Bash
#!/usr/bin/env bash
|
|
# coolify-autostart.sh
|
|
# -----------------------------------------------------------------------------
|
|
# Ensures the Coolify stack survives a power outage: on every host boot it
|
|
# starts LXC 102 (if not already up via onboot), waits for Docker inside the
|
|
# LXC, and makes sure the core Coolify containers + the Cloudflare tunnel are
|
|
# running. Idempotent and self-healing: safe to run any number of times.
|
|
#
|
|
# Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and
|
|
# invoked by the systemd unit coolify-autostart.service on multi-user.target.
|
|
#
|
|
# Timing note (measured on the 2026-08-07 boots): pve-guests takes ~78 s to
|
|
# start CT 102, the Docker daemon inside it only answers ~4-5 min after boot,
|
|
# and the last core container (`coolify`) starts ~7m40s after boot. Every wait
|
|
# here is therefore wall-clock based and generously sized; the systemd unit's
|
|
# TimeoutStartSec must stay above MAX_WAIT + SETTLE.
|
|
# -----------------------------------------------------------------------------
|
|
set -uo pipefail
|
|
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
|
|
|
LXC_ID="${COOLIFY_LXC:-102}"
|
|
LOG="${COOLIFY_LOG:-/var/log/coolify-autostart.log}" # override to dry-run without touching the real log
|
|
MAX_WAIT="${COOLIFY_MAX_WAIT:-600}" # wall-clock seconds to wait for docker
|
|
SETTLE="${COOLIFY_SETTLE:-180}" # grace for docker to start its own containers
|
|
PROBE_TIMEOUT="${COOLIFY_PROBE_TIMEOUT:-20}" # hard timeout for every call into the LXC
|
|
|
|
# Dependencies first, then the app, proxy and tunnel. These all normally come
|
|
# up on their own via Docker restart policies; this loop only heals the ones
|
|
# that didn't (e.g. restart=no, or a wedged start).
|
|
CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared)
|
|
|
|
failures=0
|
|
|
|
log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; }
|
|
now() { date +%s; }
|
|
|
|
# Every call into the LXC gets a hard timeout. During a cold boot the Docker
|
|
# daemon is busy starting ~50 containers and a single blocking `docker` call
|
|
# used to eat the entire budget silently, so the guardian was killed by systemd
|
|
# before it ever reached the container checks.
|
|
lxc() { timeout "$PROBE_TIMEOUT" pct exec "$LXC_ID" -- "$@"; }
|
|
|
|
# true | false | unknown (unknown = inspect failed, container may not exist yet)
|
|
container_state() { lxc docker inspect -f '{{.State.Running}}' "$1" 2>/dev/null || echo unknown; }
|
|
|
|
log "=== coolify-autostart start (LXC ${LXC_ID}) ==="
|
|
|
|
# 1. Ensure the LXC is running -------------------------------------------------
|
|
status="$(pct status "$LXC_ID" 2>/dev/null | awk '{print $2}')"
|
|
if [ "$status" != "running" ]; then
|
|
log "LXC ${LXC_ID} status='${status:-unknown}' -> starting"
|
|
if pct start "$LXC_ID"; then
|
|
log "pct start issued"
|
|
else
|
|
log "ERROR: pct start ${LXC_ID} failed"
|
|
failures=$((failures + 1))
|
|
fi
|
|
else
|
|
log "LXC ${LXC_ID} already running"
|
|
fi
|
|
|
|
# 2. Wait for the Docker daemon inside the LXC to respond ----------------------
|
|
# Wall-clock deadline (not a sleep counter) and a cheap probe: `docker
|
|
# version` hits /version, while `docker info` enumerates every container and
|
|
# plugin and stalls for minutes on a loaded daemon.
|
|
start_ts="$(now)"
|
|
deadline=$((start_ts + MAX_WAIT))
|
|
until lxc docker version --format '{{.Server.Version}}' >/dev/null 2>&1; do
|
|
if [ "$(now)" -ge "$deadline" ]; then
|
|
log "ERROR: docker not ready after ${MAX_WAIT}s (wall clock) -> aborting"
|
|
exit 1
|
|
fi
|
|
sleep 5
|
|
done
|
|
log "docker ready after $(( $(now) - start_ts ))s"
|
|
|
|
# 3. Ensure the core Coolify containers + tunnel container are running ---------
|
|
# Docker's own restart policies bring these up over several minutes, so poll
|
|
# each one until SETTLE expires before forcing a start.
|
|
settle_deadline=$(($(now) + SETTLE))
|
|
for c in "${CORE_CONTAINERS[@]}"; do
|
|
st="$(container_state "$c")"
|
|
while [ "$st" != "true" ] && [ "$(now)" -lt "$settle_deadline" ]; do
|
|
sleep 5
|
|
st="$(container_state "$c")"
|
|
done
|
|
case "$st" in
|
|
true) log "container ${c}: running" ;;
|
|
*)
|
|
log "container ${c}: running=${st} -> starting"
|
|
if lxc docker start "$c" >/dev/null 2>&1; then
|
|
log "started ${c}"
|
|
elif [ "$st" = "unknown" ]; then
|
|
log "WARN container ${c}: not found (skipping)"
|
|
else
|
|
log "ERROR: could not start ${c}"
|
|
failures=$((failures + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# 4. Ensure the systemd cloudflared tunnel inside the LXC is up ----------------
|
|
# (second connector to the same tunnel; belt-and-suspenders alongside the
|
|
# cloudflared Docker container above.)
|
|
if lxc systemctl is-enabled cloudflared >/dev/null 2>&1; then
|
|
if ! lxc systemctl start cloudflared >/dev/null 2>&1; then
|
|
log "WARN: cloudflared.service start returned non-zero"
|
|
fi
|
|
log "cloudflared.service is-active=$(lxc systemctl is-active cloudflared 2>/dev/null || echo unknown)"
|
|
else
|
|
log "WARN: cloudflared.service not enabled inside LXC ${LXC_ID}"
|
|
fi
|
|
|
|
# 5. Prove the tunnel actually reached Cloudflare this boot --------------------
|
|
# Without this the log said "ensured up" even when no connector registered.
|
|
conns="$(lxc journalctl -u cloudflared -b --no-pager 2>/dev/null | grep -c 'Registered tunnel connection' || true)"
|
|
conns="${conns//[!0-9]/}" # grep -c exits 1 on zero matches; keep only the digits
|
|
conns="${conns:-0}"
|
|
log "cloudflared.service registered tunnel connections this boot: ${conns}"
|
|
if [ "$conns" -eq 0 ]; then
|
|
log "WARN: no tunnel connection registered yet (edge may still be connecting)"
|
|
fi
|
|
|
|
log "=== coolify-autostart done (failures=${failures}) ==="
|
|
[ "$failures" -eq 0 ] || exit 1
|