Actualiza toolkit operativo y documentación

This commit is contained in:
urieljareth
2026-09-10 20:53:50 -06:00
parent 3b7209dcc1
commit 714057bfc8
69 changed files with 6023 additions and 384 deletions
+62 -17
View File
@@ -8,20 +8,40 @@
#
# Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and
# invoked by the systemd unit coolify-autostart.service on multi-user.target.
#
# Timing note (measured on the 2026-08-07 boots): pve-guests takes ~78 s to
# start CT 102, the Docker daemon inside it only answers ~4-5 min after boot,
# and the last core container (`coolify`) starts ~7m40s after boot. Every wait
# here is therefore wall-clock based and generously sized; the systemd unit's
# TimeoutStartSec must stay above MAX_WAIT + SETTLE.
# -----------------------------------------------------------------------------
set -uo pipefail
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
LXC_ID="${COOLIFY_LXC:-102}"
LOG="/var/log/coolify-autostart.log"
MAX_WAIT="${COOLIFY_MAX_WAIT:-180}" # seconds to wait for docker inside the LXC
LOG="${COOLIFY_LOG:-/var/log/coolify-autostart.log}" # override to dry-run without touching the real log
MAX_WAIT="${COOLIFY_MAX_WAIT:-600}" # wall-clock seconds to wait for docker
SETTLE="${COOLIFY_SETTLE:-180}" # grace for docker to start its own containers
PROBE_TIMEOUT="${COOLIFY_PROBE_TIMEOUT:-20}" # hard timeout for every call into the LXC
# Dependencies first, then the app, proxy and tunnel. These all normally come
# up on their own via Docker restart policies; this loop only heals the ones
# that didn't (e.g. restart=no, or a wedged start).
CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared)
failures=0
log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; }
now() { date +%s; }
# Every call into the LXC gets a hard timeout. During a cold boot the Docker
# daemon is busy starting ~50 containers and a single blocking `docker` call
# used to eat the entire budget silently, so the guardian was killed by systemd
# before it ever reached the container checks.
lxc() { timeout "$PROBE_TIMEOUT" pct exec "$LXC_ID" -- "$@"; }
# true | false | unknown (unknown = inspect failed, container may not exist yet)
container_state() { lxc docker inspect -f '{{.State.Running}}' "$1" 2>/dev/null || echo unknown; }
log "=== coolify-autostart start (LXC ${LXC_ID}) ==="
@@ -33,35 +53,48 @@ if [ "$status" != "running" ]; then
log "pct start issued"
else
log "ERROR: pct start ${LXC_ID} failed"
failures=$((failures + 1))
fi
else
log "LXC ${LXC_ID} already running"
fi
# 2. Wait for the Docker daemon inside the LXC to respond ----------------------
waited=0
until pct exec "$LXC_ID" -- docker info >/dev/null 2>&1; do
if [ "$waited" -ge "$MAX_WAIT" ]; then
log "ERROR: docker not ready after ${MAX_WAIT}s -> aborting"
# Wall-clock deadline (not a sleep counter) and a cheap probe: `docker
# version` hits /version, while `docker info` enumerates every container and
# plugin and stalls for minutes on a loaded daemon.
start_ts="$(now)"
deadline=$((start_ts + MAX_WAIT))
until lxc docker version --format '{{.Server.Version}}' >/dev/null 2>&1; do
if [ "$(now)" -ge "$deadline" ]; then
log "ERROR: docker not ready after ${MAX_WAIT}s (wall clock) -> aborting"
exit 1
fi
sleep 5
waited=$((waited + 5))
done
log "docker ready after ${waited}s"
log "docker ready after $(( $(now) - start_ts ))s"
# 3. Ensure the core Coolify containers + tunnel container are running ---------
# Docker's own restart policies bring these up over several minutes, so poll
# each one until SETTLE expires before forcing a start.
settle_deadline=$(($(now) + SETTLE))
for c in "${CORE_CONTAINERS[@]}"; do
st="$(pct exec "$LXC_ID" -- docker inspect -f '{{.State.Running}}' "$c" 2>/dev/null || echo missing)"
st="$(container_state "$c")"
while [ "$st" != "true" ] && [ "$(now)" -lt "$settle_deadline" ]; do
sleep 5
st="$(container_state "$c")"
done
case "$st" in
true) log "container ${c}: running" ;;
missing) log "WARN container ${c}: not found (skipping)" ;;
true) log "container ${c}: running" ;;
*)
log "container ${c}: running=${st} -> starting"
if pct exec "$LXC_ID" -- docker start "$c" >/dev/null 2>&1; then
if lxc docker start "$c" >/dev/null 2>&1; then
log "started ${c}"
elif [ "$st" = "unknown" ]; then
log "WARN container ${c}: not found (skipping)"
else
log "ERROR: could not start ${c}"
failures=$((failures + 1))
fi
;;
esac
@@ -70,12 +103,24 @@ done
# 4. Ensure the systemd cloudflared tunnel inside the LXC is up ----------------
# (second connector to the same tunnel; belt-and-suspenders alongside the
# cloudflared Docker container above.)
if pct exec "$LXC_ID" -- systemctl is-enabled cloudflared >/dev/null 2>&1; then
if pct exec "$LXC_ID" -- systemctl start cloudflared >/dev/null 2>&1; then
log "cloudflared.service ensured up"
else
if lxc systemctl is-enabled cloudflared >/dev/null 2>&1; then
if ! lxc systemctl start cloudflared >/dev/null 2>&1; then
log "WARN: cloudflared.service start returned non-zero"
fi
log "cloudflared.service is-active=$(lxc systemctl is-active cloudflared 2>/dev/null || echo unknown)"
else
log "WARN: cloudflared.service not enabled inside LXC ${LXC_ID}"
fi
log "=== coolify-autostart done ==="
# 5. Prove the tunnel actually reached Cloudflare this boot --------------------
# Without this the log said "ensured up" even when no connector registered.
conns="$(lxc journalctl -u cloudflared -b --no-pager 2>/dev/null | grep -c 'Registered tunnel connection' || true)"
conns="${conns//[!0-9]/}" # grep -c exits 1 on zero matches; keep only the digits
conns="${conns:-0}"
log "cloudflared.service registered tunnel connections this boot: ${conns}"
if [ "$conns" -eq 0 ]; then
log "WARN: no tunnel connection registered yet (edge may still be connecting)"
fi
log "=== coolify-autostart done (failures=${failures}) ==="
[ "$failures" -eq 0 ] || exit 1