#!/usr/bin/env bash # coolify-autostart.sh # ----------------------------------------------------------------------------- # Ensures the Coolify stack survives a power outage: on every host boot it # starts LXC 102 (if not already up via onboot), waits for Docker inside the # LXC, and makes sure the core Coolify containers + the Cloudflare tunnel are # running. Idempotent and self-healing: safe to run any number of times. # # Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and # invoked by the systemd unit coolify-autostart.service on multi-user.target. # # Timing note (measured on the 2026-08-07 boots): pve-guests takes ~78 s to # start CT 102, the Docker daemon inside it only answers ~4-5 min after boot, # and the last core container (`coolify`) starts ~7m40s after boot. Every wait # here is therefore wall-clock based and generously sized; the systemd unit's # TimeoutStartSec must stay above MAX_WAIT + SETTLE. # ----------------------------------------------------------------------------- set -uo pipefail export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin LXC_ID="${COOLIFY_LXC:-102}" LOG="${COOLIFY_LOG:-/var/log/coolify-autostart.log}" # override to dry-run without touching the real log MAX_WAIT="${COOLIFY_MAX_WAIT:-600}" # wall-clock seconds to wait for docker SETTLE="${COOLIFY_SETTLE:-180}" # grace for docker to start its own containers PROBE_TIMEOUT="${COOLIFY_PROBE_TIMEOUT:-20}" # hard timeout for every call into the LXC # Dependencies first, then the app, proxy and tunnel. These all normally come # up on their own via Docker restart policies; this loop only heals the ones # that didn't (e.g. restart=no, or a wedged start). CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared) failures=0 log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; } now() { date +%s; } # Every call into the LXC gets a hard timeout. During a cold boot the Docker # daemon is busy starting ~50 containers and a single blocking `docker` call # used to eat the entire budget silently, so the guardian was killed by systemd # before it ever reached the container checks. lxc() { timeout "$PROBE_TIMEOUT" pct exec "$LXC_ID" -- "$@"; } # true | false | unknown (unknown = inspect failed, container may not exist yet) container_state() { lxc docker inspect -f '{{.State.Running}}' "$1" 2>/dev/null || echo unknown; } log "=== coolify-autostart start (LXC ${LXC_ID}) ===" # 1. Ensure the LXC is running ------------------------------------------------- status="$(pct status "$LXC_ID" 2>/dev/null | awk '{print $2}')" if [ "$status" != "running" ]; then log "LXC ${LXC_ID} status='${status:-unknown}' -> starting" if pct start "$LXC_ID"; then log "pct start issued" else log "ERROR: pct start ${LXC_ID} failed" failures=$((failures + 1)) fi else log "LXC ${LXC_ID} already running" fi # 2. Wait for the Docker daemon inside the LXC to respond ---------------------- # Wall-clock deadline (not a sleep counter) and a cheap probe: `docker # version` hits /version, while `docker info` enumerates every container and # plugin and stalls for minutes on a loaded daemon. start_ts="$(now)" deadline=$((start_ts + MAX_WAIT)) until lxc docker version --format '{{.Server.Version}}' >/dev/null 2>&1; do if [ "$(now)" -ge "$deadline" ]; then log "ERROR: docker not ready after ${MAX_WAIT}s (wall clock) -> aborting" exit 1 fi sleep 5 done log "docker ready after $(( $(now) - start_ts ))s" # 3. Ensure the core Coolify containers + tunnel container are running --------- # Docker's own restart policies bring these up over several minutes, so poll # each one until SETTLE expires before forcing a start. settle_deadline=$(($(now) + SETTLE)) for c in "${CORE_CONTAINERS[@]}"; do st="$(container_state "$c")" while [ "$st" != "true" ] && [ "$(now)" -lt "$settle_deadline" ]; do sleep 5 st="$(container_state "$c")" done case "$st" in true) log "container ${c}: running" ;; *) log "container ${c}: running=${st} -> starting" if lxc docker start "$c" >/dev/null 2>&1; then log "started ${c}" elif [ "$st" = "unknown" ]; then log "WARN container ${c}: not found (skipping)" else log "ERROR: could not start ${c}" failures=$((failures + 1)) fi ;; esac done # 4. Ensure the systemd cloudflared tunnel inside the LXC is up ---------------- # (second connector to the same tunnel; belt-and-suspenders alongside the # cloudflared Docker container above.) if lxc systemctl is-enabled cloudflared >/dev/null 2>&1; then if ! lxc systemctl start cloudflared >/dev/null 2>&1; then log "WARN: cloudflared.service start returned non-zero" fi log "cloudflared.service is-active=$(lxc systemctl is-active cloudflared 2>/dev/null || echo unknown)" else log "WARN: cloudflared.service not enabled inside LXC ${LXC_ID}" fi # 5. Prove the tunnel actually reached Cloudflare this boot -------------------- # Without this the log said "ensured up" even when no connector registered. conns="$(lxc journalctl -u cloudflared -b --no-pager 2>/dev/null | grep -c 'Registered tunnel connection' || true)" conns="${conns//[!0-9]/}" # grep -c exits 1 on zero matches; keep only the digits conns="${conns:-0}" log "cloudflared.service registered tunnel connections this boot: ${conns}" if [ "$conns" -eq 0 ]; then log "WARN: no tunnel connection registered yet (edge may still be connecting)" fi log "=== coolify-autostart done (failures=${failures}) ===" [ "$failures" -eq 0 ] || exit 1