Actualiza toolkit operativo y documentación

This commit is contained in:
urieljareth
2026-09-10 20:53:50 -06:00
parent 3b7209dcc1
commit 714057bfc8
69 changed files with 6023 additions and 384 deletions
+60 -16
View File
@@ -1,28 +1,72 @@
# Plantilla de secretos. Copia a .env.local.ps1 (gitignored) y rellena.
# Copy-Item .env.example .env.local.ps1
#
# Nota: .env.local.ps1 es un script de PowerShell que se carga con dot-source
# (`. .\.env.local.ps1`), así que cada línea va como $env:NOMBRE = "valor".
# Este archivo usa formato KEY=VALUE solo como referencia de qué se necesita.
#
# Qué necesita cada ruta: docs/TOOL-INDEX.md
# ── Proxmox: SSH ────────────────────────────────────────────────────────────
# Basta con esto para todo el trabajo por SSH (el resto son los defaults del
# repo, en scripts/ProxmoxAgent.ps1). La llave del repo (keys/proxmox_ed25519)
# es idéntica a C:\Users\Uriel Jareth\.ssh\coolify_key — sin passphrase.
# NOTA: la ruta antigua …\.openclaw\workspace\proxmox_key_win YA NO EXISTE.
PROXMOX_HOST=192.168.0.200 PROXMOX_HOST=192.168.0.200
PROXMOX_NODE=thinkcentre PROXMOX_NODE=thinkcentre
PROXMOX_USER=root PROXMOX_USER=root
PROXMOX_SSH_KEY=C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win PROXMOX_SSH_KEY=keys/proxmox_ed25519
PROXMOX_API_BASE_URL=https://192.168.0.200:8006/api2/json
PROXMOX_API_TOKEN_ID=root@pam!openclaw
PROXMOX_API_TOKEN_SECRET=REPLACE_WITH_TOKEN_SECRET
PROXMOX_COOLIFY_LXC=102 PROXMOX_COOLIFY_LXC=102
# ── Proxmox: API REST ───────────────────────────────────────────────────────
# Requerido por Invoke-ProxmoxApi. Sin ambos, lanza excepción.
# Genera el token en la UI de Proxmox: Datacenter → Permissions → API Tokens.
PROXMOX_API_BASE_URL=https://192.168.0.200:8006/api2/json
PROXMOX_API_TOKEN_ID=root@pam!nombre-del-token
PROXMOX_API_TOKEN_SECRET=REPLACE_WITH_TOKEN_SECRET
# ── Coolify: API ────────────────────────────────────────────────────────────
# Requerido por coolify_skill/scripts/Invoke-CoolifyApi.ps1.
# Instancia actual: v4.3.14 — la API está completa contra el ORIGEN
# (http://<COOLIFY_HOST_LAN>:8000/api/v1); vía Cloudflare, /applications/* y
# /github-apps devuelven 404 por un bloqueo del edge (no de Coolify).
# Desde v4.2, los endpoints de estado exigen POST (p. ej. /deploy).
COOLIFY_API_URL=https://coolify.urieljareth.org/api/v1 COOLIFY_API_URL=https://coolify.urieljareth.org/api/v1
COOLIFY_TOKEN=REPLACE_WITH_COOLIFY_TOKEN COOLIFY_TOKEN=REPLACE_WITH_COOLIFY_TOKEN
# Used by deploy_skill (New-GitHubRepo.ps1, Invoke-GitHubApi.ps1) to create # ── Coolify: login de la UI web ─────────────────────────────────────────────
# repos via the GitHub REST API. git push/pull uses Windows Credential Manager # Requerido SOLO por el flujo Playwright de deploy_skill/scripts/coolify-ui/,
# (wincred) and does NOT need this token in the URL. # que es la vía soportada para apps git build-from-source (porque la API de
# Generate at https://github.com/settings/tokens (classic 'repo' scope, or # /applications da 404). Son las credenciales con las que entras al dashboard.
# fine-grained with Contents:Read+Write and Metadata:Read). COOLIFY_EMAIL=REPLACE_WITH_COOLIFY_LOGIN_EMAIL
GITHUB_TOKEN=REPLACE_WITH_GITHUB_PAT COOLIFY_PASSWORD=REPLACE_WITH_COOLIFY_LOGIN_PASSWORD
# Used by gitea_skill (Invoke-GiteaApi.ps1, New-GiteaRepo.ps1, Sync-GiteaRemote.ps1). # ── Cloudflare: API ─────────────────────────────────────────────────────────
# Self-hosted Gitea behind the Cloudflare tunnel. Generate a token at: # Requerido por scripts/Invoke-CloudflareApi.ps1 (túnel + DNS).
# <GITEA_URL>/user/settings/applications # El túnel es gestionado desde el dashboard: corrige rutas ahí o por esta API,
# Scopes needed: read:repository, write:repository, read:user (and write:admin # nunca editando archivos en el host.
# only if you manage other users). git push uses a one-shot http.extraHeader # Genera el token en https://dash.cloudflare.com/profile/api-tokens
# injected by Sync-GiteaRemote.ps1 — the token is NOT persisted to .git/config. CLOUDFLARE_API_TOKEN=REPLACE_WITH_CLOUDFLARE_API_TOKEN
# ── GitHub ──────────────────────────────────────────────────────────────────
# Usado por deploy_skill (New-GitHubRepo.ps1, Invoke-GitHubApi.ps1) para crear
# repos vía la API REST. `git push` usa Windows Credential Manager (wincred) y
# NO necesita este token en la URL.
# Un PAT aquí puede caducar (el de 2026-08 lo hizo); el token vivo de wincred
# se recupera con:
# printf "protocol=https\nhost=github.com\n\n" | git credential fill
# Genera en https://github.com/settings/tokens — scope clásico 'repo', o
# fine-grained con Contents:Read+Write y Metadata:Read.
GITHUB_TOKEN=REPLACE_WITH_GITHUB_PAT
GITHUB_OWNER=urieljarethbusiness-cpu
# ── Gitea ───────────────────────────────────────────────────────────────────
# Usado por gitea_skill (Invoke-GiteaApi.ps1, New-GiteaRepo.ps1,
# Sync-GiteaRemote.ps1). Gitea self-hosted detrás del túnel de Cloudflare.
# Genera un token en <GITEA_URL>/user/settings/applications
# Scopes: read:repository, write:repository, read:user (write:admin solo si
# administras otros usuarios). `git push` usa un http.extraHeader de un solo uso
# inyectado por Sync-GiteaRemote.ps1 — el token NO se persiste en .git/config.
GITEA_URL=https://gitea-hjwh0svsoo9p5w5kj2j6b1bd.urieljareth.org GITEA_URL=https://gitea-hjwh0svsoo9p5w5kj2j6b1bd.urieljareth.org
GITEA_USER=urieljareth GITEA_USER=urieljareth
GITEA_TOKEN=REPLACE_WITH_GITEA_TOKEN GITEA_TOKEN=REPLACE_WITH_GITEA_TOKEN
+9
View File
@@ -7,6 +7,9 @@ ACCESS.md
*.pem *.pem
*.pfx *.pfx
*.crt *.crt
keys/
.keys/
.ssh/
*.log *.log
*.tmp *.tmp
__pycache__/ __pycache__/
@@ -20,3 +23,9 @@ node_modules/
.claude/ .claude/
.opencode/ .opencode/
# Copias de rollback de composes de Coolify (pueden contener valores reales)
backups/
# homelab-skill: bundle autocontenido con credenciales reales para agentes (nunca commitear)
homelab-skill/
+164 -99
View File
@@ -1,128 +1,193 @@
# CLAUDE.md # CLAUDE.md
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. Guía para Claude Code (claude.ai/code) al trabajar en este repositorio.
## What this repo is ## Antes de invocar cualquier herramienta
This is **not an application codebase**. It is an operations toolkit: PowerShell 1. Carga los secretos: `. .\.env.local.ps1` (gitignored). Sin esto, todas las
wrapper scripts plus context/runbook documentation that let an agent diagnose and llamadas a API lanzan excepción.
manage a single local Proxmox host (`192.168.0.200`, node `thinkcentre`) and the 2. Lee **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** — es el catálogo canónico de
self-hosted Coolify stack running inside its LXC `102`. There is no build, lint, todo lo ejecutable: firma real de cada script, variables de entorno que
or test step — the "commands" are the operational scripts in `scripts/` and requiere, y si es solo-lectura o mutante. **Su §1 lista seis gotchas que
`coolify_skill/scripts/`. producen resultados silenciosamente incorrectos.** No los adivines.
3. Para el estado actual del sistema (qué existe, qué versión corre):
[docs/proxmox-inventory.md](docs/proxmox-inventory.md).
User-facing docs are in Spanish; scripts and skill files are in English. Los seis gotchas, en una línea cada uno (detalle en el índice):
## Core architecture - **`-Raw` está invertido entre wrappers.** En Coolify y Cloudflare, sin `-Raw`
recibes un *string*, no objetos: filtrar da vacío sin error.
- **`Invoke-ProxmoxSsh.ps1` corrompe comillas anidadas.** Para comandos con más
de un nivel de comillas, codifica en base64.
- **Los nombres de contenedor llevan sufijo uuid y cambian en cada redeploy.**
Resuélvelos siempre; nunca los escribas a mano.
- **`/applications/*` de la API de Coolify da 404 por el hostname público
(Cloudflare), no por Coolify.** Contra el origen
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`) la API
completa responde; y desde v4.2 los endpoints de estado exigen POST.
- **"Verde en Coolify" no es "enrutado en Traefik".** La UI mira `running`;
Traefik exige `healthy`. Un servicio nuevo aún arrancando da `503 no available
server` sin que nada esté mal configurado.
- **Un `502` casi siempre es el puerto.** Coolify saca el puerto de Traefik del
`:puerto` del FQDN guardado; sin él cae al `EXPOSE` de la imagen. Si no
coinciden, `connection refused`.
**Everything reaches the host through one SSH path.** There is no direct Docker or ## Qué es este repo
local network access. The layering is:
**No es el código de una aplicación.** Es un toolkit de operaciones: wrappers de
PowerShell más documentación de contexto y runbooks que permiten a un agente
diagnosticar y administrar un homelab — un host Proxmox local (`192.168.0.200`,
nodo `thinkcentre`) y el stack Coolify que corre dentro de su LXC `102`. No hay
build, lint ni tests: los "comandos" son los scripts operativos.
**Convención de idioma:** la documentación (`docs/`, `README.md`, este archivo)
está en español. Los scripts y los `SKILL.md`/`TOOLS.md` de cada skill están en
inglés.
## Router de intención → herramienta
Lo que pide el usuario, y con qué se resuelve. Para la firma completa de cada
script, ve al [índice](docs/TOOL-INDEX.md).
| El usuario pide… | Usa |
|---|---|
| "¿está bien el servidor?", "revisa el Proxmox" | `.\scripts\Test-ProxmoxConnection.ps1` y luego `.\scripts\Get-ProxmoxInventory.ps1` |
| "inventario de LXC/VMs", "qué hay corriendo" | `.\scripts\Get-ProxmoxInventory.ps1` |
| "¿está corriendo X?", "estado de los contenedores" | `.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -Filter <regex>` |
| "logs de X" | `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker logs <nombre-resuelto> --tail 100"` |
| "qué apps hay en Coolify", "dame el uuid de X" | `.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw` |
| cualquier cosa de la API de Coolify | `.\coolify_skill\scripts\Invoke-CoolifyApi.ps1` — pero revisa primero qué endpoints viven (§1.4 del índice) |
| "¿está online el sitio X?" | `.\deploy_skill\scripts\Test-ServiceOnline.ps1 -Fqdn https://x.urieljareth.org` |
| "el servicio nuevo está en verde pero da 503 / `no available server`" | `.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600` — **espera, no redeployes**; ver §1.5 del índice |
| "arregla el túnel de Cloudflare", rutas/DNS | `.\scripts\Invoke-CloudflareApi.ps1` + [docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md) |
| "Chatwoot perdió el enterprise" | diagnostica con `.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep`; repara con `Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures` |
| "que arranque solo tras un apagón" | `.\scripts\Install-CoolifyAutostart.ps1 -VerifyOnly` primero |
| "publica este proyecto en Coolify" | **lee §4 del índice antes**. Stack compose → `New-CoolifyService.ps1` (arreglado y verificado el 2026-08-24); app git → flujo UI con Playwright |
| "una app de Coolify sigue el `main` de upstream y se rompió" | [docs/casos/firecrawl-stack-minimo.md](docs/casos/firecrawl-stack-minimo.md) — cambiar a imágenes precompiladas y compose propio |
| "valida que este proyecto se puede deployar" | `.\deploy_skill\scripts\Test-PreDeployChecklist.ps1 -Path <ruta> -Strict` |
| "estoy construyendo una app para este Coolify" | [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) (contrato de red, puertos, dominios, volúmenes) |
| "crea/lista un repo en Gitea", "sube esto a Gitea" | `.\gitea_skill\scripts\` — ver §5 del índice |
| "diagnosticar/iniciar Hermes (LXC 100)", "MiniMax-M3" | `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"` + [docs/casos/hermes-minimax-m3-setup.md](docs/casos/hermes-minimax-m3-setup.md) |
| "un contenedor no conecta a su base de datos" | ver "Gotcha de red Docker" abajo |
## Arquitectura
**Todo llega al host por un único camino SSH.** No hay acceso directo a Docker ni
a la red interna:
``` ```
PowerShell script → Invoke-ProxmoxSshCommand (scripts/ProxmoxAgent.ps1) script PowerShell → Invoke-ProxmoxSshCommand (scripts/ProxmoxAgent.ps1)
→ ssh [email protected] → ssh [email protected]
→ pct exec 102 -- docker ... (for any Docker/Coolify container work) → pct exec 102 -- docker ... (cualquier trabajo de Docker/Coolify)
``` ```
- [scripts/ProxmoxAgent.ps1](scripts/ProxmoxAgent.ps1) is the shared library. **Dot-source it** (`. .\scripts\ProxmoxAgent.ps1`) to get `Get-ProxmoxConfig`, `Invoke-ProxmoxSshCommand`, and `Invoke-ProxmoxApi`. Every other script dot-sources it rather than reimplementing connection logic. - [scripts/ProxmoxAgent.ps1](scripts/ProxmoxAgent.ps1) es la librería compartida.
- **Config resolution:** `Get-ProxmoxConfig` reads env vars (`PROXMOX_HOST`, `PROXMOX_NODE`, `PROXMOX_SSH_KEY`, `PROXMOX_COOLIFY_LXC`, etc.) and falls back to hardcoded local defaults. SSH works with defaults alone; **API calls require `PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`** (otherwise `Invoke-ProxmoxApi` throws). The Coolify LXC ID (`102`) comes from config — don't hardcode it in new scripts; use `$config.CoolifyLxc`. **Hazle dot-source** (`. .\scripts\ProxmoxAgent.ps1`) para obtener
- Two independent APIs: the **Proxmox REST API** (via `curl.exe -k`, PVE token header) and the **Coolify API** (via `Invoke-RestMethod`, Bearer token, base `https://coolify.urieljareth.org/api/v1`). They use different scripts and different env vars. `Get-ProxmoxConfig`, `Invoke-ProxmoxSshCommand` e `Invoke-ProxmoxApi`. Todos los
- Docker is **not** managed on the Proxmox host directly — it lives inside LXC `102`. Any container command must be wrapped as `pct exec 102 -- docker ...`. demás scripts la consumen en lugar de reimplementar la conexión.
- **The deploy pipeline** ([deploy_skill/](deploy_skill/)) is the end-to-end path from "local project" to "live on `*.urieljareth.org`": scaffold compliant Dockerfile/compose → pre-deploy checklist → create GitHub repo → push → register + deploy via Coolify API → verify. It also covers rollback. Requires `GITHUB_TOKEN` + `COOLIFY_TOKEN` in `.env.local.ps1`; `git push` uses Windows Credential Manager (`wincred`), verified for the `urieljarethbusiness-cpu` GitHub account. See [deploy_skill/SKILL.md](deploy_skill/SKILL.md) and [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md). - **Resolución de configuración:** `Get-ProxmoxConfig` lee variables de entorno
(`PROXMOX_HOST`, `PROXMOX_NODE`, `PROXMOX_SSH_KEY`, `PROXMOX_COOLIFY_LXC`…) y
cae a los defaults locales hardcodeados. SSH funciona solo con los defaults;
**las llamadas a la API exigen `PROXMOX_API_TOKEN_ID` +
`PROXMOX_API_TOKEN_SECRET`** (si faltan, `Invoke-ProxmoxApi` lanza). El ID del
LXC de Coolify (`102`) viene de la config: usa `$config.CoolifyLxc`, no lo
hardcodees en scripts nuevos.
- **Cuatro APIs independientes**, con scripts y variables distintas: Proxmox REST
(`curl.exe -k`, header de token PVE), Coolify (`Invoke-RestMethod`, Bearer,
base `https://coolify.urieljareth.org/api/v1`), Cloudflare y Gitea.
- Docker **no** se administra en el host Proxmox: vive dentro del LXC `102`.
Todo comando de contenedor va envuelto en `pct exec 102 -- docker ...`.
## Common commands ## Gotcha de red Docker
Run from the project root in PowerShell. Cuando una app de Coolify y su base de datos son contenedores hermanos en la
misma red Docker, la app debe alcanzar la DB **por su nombre de servicio Docker,
no por `localhost`** (Nextcloud, por ejemplo, usa el host `nextcloud-db`). El DNS
entre servicios es la causa raíz habitual de los fallos de "conexión a la base de
datos" aquí. Compruébalo con:
```powershell ```powershell
# Load private secrets (gitignored) — needed for any API call .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <app> getent hosts <nombre-servicio>"
. .\.env.local.ps1
# Smoke test: config + SSH read + Docker sample + API auth (exits 1 on any FAIL)
.\scripts\Test-ProxmoxConnection.ps1
# Full inventory snapshot (host, LXC, QEMU, Docker in LXC 102)
.\scripts\Get-ProxmoxInventory.ps1
# Arbitrary read-only SSH command
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps -a"
# Proxmox REST API (after env token loaded)
. .\scripts\ProxmoxAgent.ps1
Invoke-ProxmoxApi -Path "/version"
# Coolify container status through LXC 102
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All
# Coolify API (requires COOLIFY_TOKEN)
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects"
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/projects" -BodyJson $body
# Cloudflare API (requires CLOUDFLARE_API_TOKEN) — full tunnel/DNS control
.\scripts\Invoke-CloudflareApi.ps1 -Path "/user/tokens/verify"
# Gitea API (requires GITEA_URL + GITEA_TOKEN) — self-hosted git hosting
.\gitea_skill\scripts\Test-GiteaConnection.ps1
.\gitea_skill\scripts\Get-GiteaRepo.ps1 -List
.\gitea_skill\scripts\Invoke-GiteaApi.ps1 -Path "/user"
.\gitea_skill\scripts\Sync-GiteaRemote.ps1 -AppPath .\my-app -CreateIfMissing -Force
``` ```
For the Coolify API reference: don't read the whole tree. Search ## Reglas de operación (las imponen las skills — cúmplelas)
`coolify_skill/references/` with `rg`, then open the single matching
`references/ops/*.md` operation file.
## Deploying a new project to Coolify - **Solo-lectura primero.** El default es diagnosticar: list, status, logs,
inspect, health checks.
- **Confirma antes de cualquier cambio de estado.** Pregunta explícitamente antes
de: `pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose
up-down; deploys de Coolify o cualquier `POST`/`PUT`/`PATCH`/`DELETE`; escritura
de variables de entorno; y todo cambio de firewall, red, storage, volumen,
clave o token. Para acciones riesgosas: captura el estado actual y enuncia
primero el camino de rollback.
- **Nunca escribas secretos en el repo.** Ni tokens, passwords, claves privadas,
cookies o secretos de token PVE — no en Markdown, no en scripts, no en logs.
Viven solo en `.env.local.ps1` (gitignored) o en el almacén del SO. `.gitignore`
bloquea además `ACCESS.md`, `*.key`, `*.pem`, `*.crt`. Al depurar bases de
datos, verifica conectividad sin imprimir credenciales.
- **Prefiere los scripts del repo** antes que cadenas de comandos manuales
largas, y no ejecutes comandos destructivos amplios construidos desde strings
generados.
End-to-end pipeline (scaffold → checklist → GitHub repo → push → Coolify app + ## Chatwoot: el parche enterprise se degrada solo — pero un guard lo auto-repara
deploy → verify). Requires `GITHUB_TOKEN` and `COOLIFY_TOKEN` in
`.env.local.ps1`. `git push` uses Windows Credential Manager, not the PAT.
```powershell El parche no se pierde al actualizar. `Internal::CheckNewVersionsJob` hace ping
. .\.env.local.ps1 diario a `hub.2.chatwoot.com` (a los `MD5(INSTALLATION_IDENTIFIER).hex % 1440`
minutos pasada la medianoche UTC = **16:16 UTC** en esta instalación) y reescribe
`INSTALLATION_PRICING_PLAN` con la respuesta del hub; después
`ReconcilePlanConfigService` apaga los 9 feature flags premium en **todas** las
cuentas. De ahí tres consecuencias:
# Full pipeline on a local project - **El SQL de 3 filas no basta.** Los flags por cuenta viven en
.\deploy_skill\scripts\Publish-ProjectToCoolify.ps1 ` `accounts.feature_flags` (bitmask). Usa
-AppPath .\my-app -AppName my-app ` `.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures`;
-Fqdn https://my-app.urieljareth.org ` comprueba el estado con `.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep`.
-Stack node -AppPort 8080 -Init - **Los `UPDATE` por SQL no invalidan la caché Redis de `GlobalConfig`** (TTL de
1 día, `V1:GLOBAL_CONFIG:*`) porque se saltan el `after_commit :clear_cache` de
`InstallationConfig`. Cualquier parche por SQL debe llamar además a
`GlobalConfig.clear_cache`.
- **No bloquees `hub.2.chatwoot.com`.** Ese mismo host relaya el push móvil
(`ChatwootHub.send_push`), activo aquí porque `FIREBASE_*` está vacío. En su
lugar, [scripts/chatwoot-enterprise-guard.sh](scripts/chatwoot-enterprise-guard.sh)
corre en el host Proxmox desde `/root/scripts/`, agendado por
`/etc/cron.d/chatwoot-enterprise-guard` cada 5 minutos, y repara un revert
detectado en ~13s (log: `/var/log/chatwoot-enterprise-guard.log`, solo escribe
cuando actúa).
# Or step-by-step Nunca pulses `Refresh` en `/super_admin/settings`.
.\deploy_skill\scripts\Initialize-CoolifyProject.ps1 -Path .\my-app -Stack node
.\deploy_skill\scripts\Test-PreDeployChecklist.ps1 -Path .\my-app -Strict
.\deploy_skill\scripts\New-GitHubRepo.ps1 -Name my-app
.\deploy_skill\scripts\New-CoolifyApplication.ps1 -RepoUrl https://github.com/urieljarethbusiness-cpu/my-app.git `
-Fqdn https://my-app.urieljareth.org -PortsExposes 8080
.\deploy_skill\scripts\Test-PostDeploy.ps1 -Fqdn https://my-app.urieljareth.org -ContainerName my-app-main
# Rollback to a previous commit > **Estado al 2026-08-07:** corre **`chatwoot/chatwoot:v4.16.2`** (el pin
.\deploy_skill\scripts\Invoke-CoolifyRollback.ps1 -AppPath .\my-app -ApplicationUuid <uuid> -CommitSha <sha> > documentado antes era `v4.16.1`, así que el pin no sostuvo la versión). El plan
``` > está en `enterprise` y una ejecución manual del guard pasa correctamente, pero
> el log registra `ERROR: no se pudo leer INSTALLATION_PRICING_PLAN` durante la
> ventana del update. El guard tiene el nombre del contenedor hardcodeado
> (`postgres-c11xzy2tx2cdapm32f5b89vy`), así que **un redeploy que cambie el
> sufijo lo deja ciego**. Análisis completo, registro de ejecución y rollback:
> [docs/runbooks/chatwoot-update.md](docs/runbooks/chatwoot-update.md).
## Operating rules (enforced by the skills — follow them) ## Túnel de Cloudflare
- **Read-only first.** Default to diagnostics: list, status, logs, inspect, health checks. El túnel es **gestionado desde el dashboard** (el ingress baja del edge — se ve
- **Confirm before any state change.** Explicitly ask the user before: `pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose up/down; Coolify deploys or any `POST`/`PUT`/`PATCH`/`DELETE`; env-var writes; and any firewall/network/storage/volume/key/token change. For risky actions, capture current state and state the rollback path first. como `INF Updated to new configuration version=N` en los logs de `cloudflared`).
- **Never write secrets into the repo.** No tokens, passwords, private keys, cookies, or PVE token secrets in Markdown, scripts, or logs. Secrets live only in `.env.local.ps1` (gitignored) or the OS secret store. `.gitignore` also blocks `ACCESS.md`, `*.key`, `*.pem`, `*.crt`, etc. When debugging databases, verify connectivity without echoing credentials. Corrige rutas en el dashboard o vía
- **Prefer repo scripts over long manual command strings**, and don't run broad destructive commands built from generated strings. [scripts/Invoke-CloudflareApi.ps1](scripts/Invoke-CloudflareApi.ps1), **nunca**
editando archivos en el host. Los puertos 6001/6002 deben usar `http://`. Ver
[docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md).
## Docker networking gotcha ## Dónde vive el contexto
When a Coolify app and its database are sibling containers on the same Docker | Qué | Dónde |
network, the app must reach the DB by its **Docker service name, not `localhost`** |---|---|
(e.g. Nextcloud uses host `nextcloud-db`). Service-to-service DNS is the usual | **Catálogo de herramientas + gotchas** | **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** |
root cause of "database connection" failures here — check with | Estado verificado del sistema | [docs/proxmox-inventory.md](docs/proxmox-inventory.md) |
`pct exec 102 -- docker exec <app> getent hosts <service-name>`. | Procedimientos vigentes | [docs/runbooks/](docs/runbooks/) — `conexion.md`, `diagnostico.md`, `coolify-docker.md`, `seguridad.md`, `cloudflare-tunnel.md`, `autostart-coolify.md`, `nextcloud.md`, `baserow.md`, `chatwoot-update.md` |
| Casos resueltos paso a paso | [docs/casos/](docs/casos/) — `chatwoot-enterprise-patch.md`, `hermes-minimax-m3-setup.md`, `coolify-servicio-nuevo-503-no-available-server.md`, `firecrawl-stack-minimo.md` |
| Incidentes archivados (**no fuente de verdad**) | [docs/incidentes/](docs/incidentes/) |
| Contrato para *construir* una app deployable | [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) |
| Reglas operativas por dominio | [agent/SKILL.md](agent/SKILL.md), [coolify_skill/SKILL.md](coolify_skill/SKILL.md), [deploy_skill/SKILL.md](deploy_skill/SKILL.md), [gitea_skill/SKILL.md](gitea_skill/SKILL.md) |
| Ejemplos ejecutables por dominio | el `TOOLS.md` de cada carpeta `*_skill/` |
| Referencia de la API de Coolify | `coolify_skill/references/ops/*.md` — **busca con `rg`, no cargues el árbol** |
## Where context lives Para la referencia de la API de Coolify: no leas el árbol completo. Busca en
`coolify_skill/references/` con `rg` y abre el único `references/ops/*.md` que
- [docs/proxmox-inventory.md](docs/proxmox-inventory.md) — verified topology, LXC list, observed containers, access model. Treat as the source of truth for current state. coincida.
- [docs/runbooks/](docs/runbooks/) — concrete procedures: `conexion.md`, `diagnostico.md`, `coolify-docker.md`, `seguridad.md`, `cloudflare-tunnel.md`, `autostart-coolify.md` (power-outage auto-start of LXC 102 + tunnel), plus app-specific `nextcloud.md` and `baserow.md`.
- Cloudflare tunnel: the tunnel is **dashboard-managed** (ingress comes from the edge, see `INF Updated to new configuration version=N` in `cloudflared` logs) — fix routes in the dashboard or via [scripts/Invoke-CloudflareApi.ps1](scripts/Invoke-CloudflareApi.ps1), not by editing files on the host. Ports 6001/6002 must use `http://`. See [docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md).
- [agent/SKILL.md](agent/SKILL.md) + [agent/TOOLS.md](agent/TOOLS.md) — Proxmox agent operating skill.
- [coolify_skill/SKILL.md](coolify_skill/SKILL.md) + [coolify_skill/TOOLS.md](coolify_skill/TOOLS.md) — Coolify agent operating skill and API reference index.
- [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) — compatibility guide for agents **developing** an app to deploy on this Coolify instance (networking, ports, domains/TLS, env, volumes, healthchecks) + pre-deploy checklist. Source this when building a new app, not when operating an existing one.
- [deploy_skill/SKILL.md](deploy_skill/SKILL.md) + [deploy_skill/TOOLS.md](deploy_skill/TOOLS.md) — deploy pipeline skill (scaffold → push → Coolify API → verify → rollback).
- [gitea_skill/SKILL.md](gitea_skill/SKILL.md) + [gitea_skill/TOOLS.md](gitea_skill/TOOLS.md) — self-hosted Gitea skill: list/create/mirror repos, wire a `gitea` remote, push headlessly via one-shot `http.extraHeader` (token never persisted). Distinct from deploy_skill (which targets Coolify); this manages the git hosting layer.
- `PROXMOX/`, `proxmox-agent/`, `proxmox-skill/` are **legacy pointer folders** — they only redirect to the live docs above. Don't add content there.
-14
View File
@@ -1,14 +0,0 @@
# Migrado
Este archivo queda solo como puntero legacy.
La documentacion viva esta en:
- `../README.md`
- `../agent/SKILL.md`
- `../agent/TOOLS.md`
- `../docs/proxmox-inventory.md`
- `../docs/runbooks/`
Los secretos que antes estaban en Markdown deben vivir en variables de entorno
o en un `.env.local.ps1` privado.
+93 -58
View File
@@ -1,80 +1,115 @@
# Proxmox & Coolify Manager # Proxmox & Coolify Manager
Proyecto local para operar Proxmox con ayuda agentica desde Codex. Toolkit para operar un homelab con ayuda de un agente: un host Proxmox local y el
stack Coolify que corre dentro de su LXC `102`, más el túnel de Cloudflare y los
repos en GitHub y Gitea.
La idea practica es simple: este repo guarda el contexto, los runbooks y los La idea es simple: este repo guarda el contexto, los runbooks y los scripts
scripts seguros para que Codex pueda diagnosticar y ayudarte a gestionar el host seguros para que un agente pueda diagnosticar y ayudarte a gestionar la
Proxmox local sin depender de memoria suelta ni de secretos pegados en Markdown. infraestructura sin depender de memoria suelta ni de secretos pegados en Markdown.
## Empieza por aquí
| Si quieres… | Lee |
|---|---|
| saber **qué script usar** para algo | **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** — el catálogo canónico |
| saber **qué existe y qué versión corre** | [docs/proxmox-inventory.md](docs/proxmox-inventory.md) |
| que un agente opere el sistema | [CLAUDE.md](CLAUDE.md) — arquitectura + router de intención |
| resolver un problema concreto | [docs/runbooks/](docs/runbooks/) |
> **Cuatro trampas conocidas** producen resultados silenciosamente incorrectos
> (el flag `-Raw` invertido entre wrappers, comillas anidadas corrompidas en SSH,
> nombres de contenedor no adivinables, y el 404 que Cloudflare impone a
> `/applications/*` en el hostname público). Están documentadas en la §1 del
> índice. Léela antes de operar.
## Estado verificado ## Estado verificado
Verificado el 2026-05-30 desde esta maquina: Verificado el **2026-08-29** desde esta máquina:
- Host Proxmox: `192.168.0.200` - Host Proxmox: `192.168.0.200`, nodo `thinkcentre`, Proxmox VE `9.1.1`,
- Nodo: `thinkcentre` kernel `6.17.2-1-pve`
- Version: Proxmox VE `9.1.1` - LXC: `100 hermes` (running — commit `5d3c15aaa`, MiniMax-M3), `102 coolify` (running)
- Kernel: `6.17.2-1-pve` - Sin VMs QEMU
- LXC detectados: `100 hermes`, `102 coolify` - Docker corre dentro del LXC `102`: **68 contenedores**, 28 recursos
- Docker corre dentro del LXC `102` registrados en Coolify
- SSH con clave local funciona - Coolify `v4.3.14` — su API está **completa contra el origen**
- API REST autenticada funciona cuando el token se carga desde entorno (`http://192.168.0.117:8000/api/v1`); vía Cloudflare `/applications/*` y
`/github-apps` devuelven 404 (bloqueo del edge, no de Coolify). Los endpoints
de estado exigen POST desde v4.2.
- SSH con clave local: OK · API de Coolify: OK · **API de Proxmox: OK**
(token `root@pam!openclaw` cargado en `.env.local.ps1`)
- API de Cloudflare: **sin token cargado**, se gestiona desde el dashboard
Credenciales y llaves SSH verificadas: `ACCESS.md` (local, gitignored).
## Estructura ## Estructura
- `agent/SKILL.md`: reglas operativas para que Codex actue como agente Proxmox. ```
- `agent/TOOLS.md`: comandos seguros y patrones de uso. CLAUDE.md Arquitectura, router de intención y reglas para el agente
- `coolify_skill/`: skill local para operar Coolify, Docker en LXC `102` y API docs/
de Coolify sin guardar secretos. TOOL-INDEX.md Catálogo canónico de herramientas + gotchas ← empieza aquí
- `deploy_skill/`: skill para deployar proyectos nuevos a Coolify de punta a proxmox-inventory.md Estado verificado del sistema (fuente de verdad)
punta (scaffold Dockerfile/compose → checklist → repo GitHub → push → alta AGENTS-coolify-apps.md Contrato para construir una app deployable en este Coolify
via API → deploy → verificacion → rollback). Necesita `GITHUB_TOKEN` y runbooks/ Procedimientos vigentes
`COOLIFY_TOKEN` en `.env.local.ps1`. casos/ Casos resueltos paso a paso
- `gitea_skill/`: skill para operar la instancia Gitea self-hosted (crear/listar/ incidentes/ Archivo histórico — NO es fuente de verdad
buscar repos, migrar desde GitHub, wire de un remoto `gitea` y push headless scripts/ Wrappers de host: SSH, API, inventario, Cloudflare, Chatwoot
con token inyectado por invocacion — sin persistirlo en `.git/config`). apps/ Scripts de deploy específicos de una app
Necesita `GITEA_URL`, `GITEA_USER`, `GITEA_TOKEN` en `.env.local.ps1`. host/ Artefactos que se despliegan en el host Proxmox
- `docs/proxmox-inventory.md`: inventario verificado y notas de arquitectura. agent/ Skill de operación del host Proxmox
- `docs/runbooks/`: procedimientos concretos para conexion, diagnostico y Coolify. coolify_skill/ Skill de operación de Coolify + referencia de su API
- `docs/runbooks/nextcloud.md`: recuperacion y fix HTTPS para Nextcloud. deploy_skill/ Skill de deploy de proyectos nuevos a Coolify
- `docs/runbooks/baserow.md`: puesta en vivo de Baserow y fix red/Traefik. gitea_skill/ Skill de la capa de hosting git self-hosted
- `docs/casos/`: casos verificados paso a paso (ej. `chatwoot-enterprise-patch.md`).
- `docs/AGENTS-coolify-apps.md`: guia para agentes/LLMs que desarrollan una app destinada a esta instancia de Coolify (reglas de red, puertos, dominios/TLS, env, volumenes, healthchecks) + checklist pre-deploy.
- `scripts/`: wrappers PowerShell para SSH, API e inventario.
- `PROXMOX/`, `proxmox-agent/`, `proxmox-skill/`: carpetas legacy que ahora apuntan a la documentacion viva.
## Configuracion local
Los secretos no viven en el repo. Usa variables de entorno o un archivo privado
ignorado por Git, por ejemplo `.env.local.ps1`.
Plantilla:
```powershell
.\scripts\Set-ProxmoxEnv.example.ps1
``` ```
Prueba de conexion: Cada carpeta `*_skill/` tiene un `SKILL.md` (reglas operativas) y un `TOOLS.md`
(ejemplos ejecutables).
### Runbooks
| Runbook | Qué cubre |
|---|---|
| [conexion.md](docs/runbooks/conexion.md) | Establecer y verificar el acceso |
| [diagnostico.md](docs/runbooks/diagnostico.md) | Triage general del host |
| [coolify-docker.md](docs/runbooks/coolify-docker.md) | Operar Docker dentro del LXC 102 |
| [seguridad.md](docs/runbooks/seguridad.md) | Postura de seguridad y manejo de secretos |
| [cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md) | Túnel y rutas (**gestionado desde el dashboard**) |
| [autostart-coolify.md](docs/runbooks/autostart-coolify.md) | Auto-arranque del stack tras un corte de luz |
| [chatwoot-update.md](docs/runbooks/chatwoot-update.md) | Actualizar Chatwoot sin perder la edición enterprise |
| [nextcloud.md](docs/runbooks/nextcloud.md) | Recuperación y fix HTTPS |
| [baserow.md](docs/runbooks/baserow.md) | Puesta en vivo y fix de red/Traefik |
## Configuración local
Los secretos no viven en el repo. Van en `.env.local.ps1`, ignorado por Git.
Copia la plantilla y rellénala:
```powershell ```powershell
.\scripts\Test-ProxmoxConnection.ps1 Copy-Item .env.example .env.local.ps1
``` ```
Inventario rapido: Luego, en cada sesión:
```powershell
.\scripts\Get-ProxmoxInventory.ps1
```
Comando SSH puntual:
```powershell ```powershell
. .\.env.local.ps1
.\scripts\Test-ProxmoxConnection.ps1 # smoke test; sale 1 si algo falla
.\scripts\Get-ProxmoxInventory.ps1 # inventario completo
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list"
``` ```
## Reglas de operacion Para la verificación operativa con navegador (`Test-ServiceOnline.ps1`) hace
falta Node: `npm install` una vez en la raíz del repo.
- Primero diagnostico de solo lectura. ## Reglas de operación
- Cambios destructivos requieren confirmacion explicita: borrar, reiniciar,
apagar, editar red, mover discos, actualizar paquetes o modificar servicios. - **Solo-lectura primero.** El default es diagnosticar.
- Preferir scripts del repo antes que comandos manuales largos. - **Confirmación explícita antes de cualquier cambio de estado**: borrar,
- No registrar tokens, passwords ni claves privadas en Markdown. reiniciar, apagar, deployar, editar red/storage, actualizar paquetes o
modificar servicios. Para acciones riesgosas, captura el estado actual y
enuncia el rollback antes de actuar.
- **Prefiere los scripts del repo** antes que comandos manuales largos.
- **Nunca registres tokens, passwords ni claves privadas en Markdown.**
La versión completa y vinculante está en la §6 del
[índice de herramientas](docs/TOOL-INDEX.md).
+17 -8
View File
@@ -4,18 +4,27 @@ Use this project-local skill when helping manage the local Proxmox host.
## Scope ## Scope
- Host: `192.168.0.200` - Host: `192.168.0.200`, Proxmox VE `9.1.1`, single node
- Node: `thinkcentre` - Node: `thinkcentre`
- Main Docker LXC: `102` (`coolify`) - Main Docker LXC: `102` (`coolify`) — **running**, 68 containers
- Secondary LXC currently observed: `100` (`hermes`) - Secondary LXC: `100` (`hermes`) — **running** (commit `5d3c15aaa`, MiniMax-M3, IPs: `192.168.3.23` / `192.168.3.15`)
- Access methods: SSH first, Proxmox REST API when token env vars are present. - No QEMU VMs
- Access methods: SSH first, Proxmox REST API when token env vars are present
(they are **not** loaded today — see [`TOOLS.md`](TOOLS.md))
## Startup routine ## Startup routine
1. Read `README.md`, `docs/proxmox-inventory.md`, and the relevant runbook. 1. Load secrets: `. .\.env.local.ps1`.
2. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work. 2. Read [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — the canonical tool
3. For fresh state, run `.\scripts\Get-ProxmoxInventory.ps1`. catalog. **Its §1 lists four gotchas that produce silently wrong results.**
4. Prefer `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "<command>"` for remote commands. The one that bites hardest here: `Invoke-ProxmoxSsh.ps1` mangles nested
quotes, so anything with two levels of quoting needs base64 (see
[`TOOLS.md`](TOOLS.md)).
3. Read [`docs/proxmox-inventory.md`](../docs/proxmox-inventory.md) for current
state, plus the relevant runbook in `docs/runbooks/`.
4. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
5. For fresh state, run `.\scripts\Get-ProxmoxInventory.ps1`.
6. Prefer `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "<command>"` for remote commands.
## Safety policy ## Safety policy
+107 -15
View File
@@ -1,26 +1,34 @@
# Tooling # Tooling — Proxmox host
All commands assume PowerShell from the project root. All commands assume PowerShell from the project root.
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — verified
> signatures, env requirements, and read-only/mutating classification for every
> script in the repo. This file holds the host-level usage patterns.
## Environment ## Environment
```powershell ```powershell
. .\.env.local.ps1 . .\.env.local.ps1
``` ```
If no private env file is loaded, SSH still uses the local defaults from Without a private env file, SSH still works from the hardcoded defaults in
`scripts/ProxmoxAgent.ps1`. API calls require token env vars. `scripts/ProxmoxAgent.ps1`. API calls require token env vars.
## Smoke test > **Status of `.env.local.ps1` (2026-08-29):** `PROXMOX_API_TOKEN_ID`
> (`root@pam!openclaw`, verified 200 against `/version` and
> `/cluster/resources`), `PROXMOX_API_TOKEN_SECRET`, `COOLIFY_EMAIL`,
> `COOLIFY_PASSWORD` (probable — unverified) and a working `GITHUB_TOKEN`
> (extracted from Windows Credential Manager after the old PAT expired) are all
> loaded. The Proxmox REST API, the Coolify API and GitHub work. Only
> `CLOUDFLARE_API_TOKEN` is still missing — tunnel changes go through the
> Cloudflare dashboard.
## Smoke test and inventory
```powershell ```powershell
.\scripts\Test-ProxmoxConnection.ps1 .\scripts\Test-ProxmoxConnection.ps1 # config + SSH + Docker sample + API auth; exits 1 on FAIL
``` .\scripts\Get-ProxmoxInventory.ps1 # host, LXC, QEMU, Docker in LXC 102
## Inventory
```powershell
.\scripts\Get-ProxmoxInventory.ps1
``` ```
## SSH wrapper ## SSH wrapper
@@ -31,7 +39,49 @@ If no private env file is loaded, SSH still uses the local defaults from
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps -a" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps -a"
``` ```
## API from a PowerShell session ### ⚠️ Nested quotes get mangled — use base64
`Invoke-ProxmoxSshCommand` passes `$Command` as a single argument to `ssh`, and
PowerShell 5.1 destroys embedded quotes when calling a native executable. A
command with two levels of quoting arrives corrupted (`bash: line 1: -c: command
not found`). Encode it instead:
```powershell
$remote = @'
pct exec 102 -- docker exec -i <db-container> bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -c "SELECT 1"'
'@
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($remote))
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "echo $b64 | base64 -d | bash 2>&1"
```
The single-quoted here-string `@'...'@` is required so PowerShell does not expand
`$POSTGRES_PASSWORD` on the Windows side. Single-level quoting
(`docker ps --format '{{.Names}}'`) works through the plain wrapper.
## Safe read-only commands
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "hostname && pveversion && uname -r && uptime"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list && qm list"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "df -h / && free -h"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pvesh get /cluster/resources"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "systemctl --failed"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format 'table {{.Names}}\t{{.Status}}\t{{.Ports}}'"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker stats --no-stream"
```
## Resolving a container name
Coolify names containers `<service>-<uuid>` plus an optional build suffix, so they
are **not guessable and change when a redeploy recreates the container**:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i <app>"
```
## Proxmox REST API from a PowerShell session
Requires `PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`; throws without them.
```powershell ```powershell
. .\scripts\ProxmoxAgent.ps1 . .\scripts\ProxmoxAgent.ps1
@@ -40,9 +90,51 @@ Invoke-ProxmoxApi -Path "/nodes"
Invoke-ProxmoxApi -Path "/cluster/resources" Invoke-ProxmoxApi -Path "/cluster/resources"
``` ```
## Cloudflare API — tunnel and DNS
Requires `CLOUDFLARE_API_TOKEN`. `GET` is read-only; everything else mutates.
`-Raw` returns objects, the default returns a JSON string.
```powershell
.\scripts\Invoke-CloudflareApi.ps1 -Path "/user/tokens/verify"
```
The tunnel is **dashboard-managed** — fix routes there or via this API, never by
editing files on the host. See [`docs/runbooks/cloudflare-tunnel.md`](../docs/runbooks/cloudflare-tunnel.md).
## Host automation
```powershell
# Power-outage auto-start of LXC 102 + tunnel. Audit without touching anything:
.\scripts\Install-CoolifyAutostart.ps1 -VerifyOnly
```
Installed on the host: `coolify-autostart.service` (systemd, **enabled**) and the
Chatwoot guard at `/root/scripts/chatwoot-enterprise-guard.sh`, scheduled by
`/etc/cron.d/chatwoot-enterprise-guard` every 5 minutes.
## Chatwoot enterprise licence
```powershell
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep # read-only diagnosis
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun -ReenableAccountFeatures # preview the repair
```
Full context: [`docs/runbooks/chatwoot-update.md`](../docs/runbooks/chatwoot-update.md).
## Per-app deploy helpers
`scripts/apps/` holds app-specific one-off deploy scripts with hardcoded uuids
and domains — e.g. `Deploy-SoloLeveling.ps1`, which builds the image on the
server to work around a private GHCR without `read:packages`. Read the header
before running one; they are **mutating** and tied to a specific resource.
## Commands that require confirmation ## Commands that require confirmation
- `pct start`, `pct shutdown`, `pct reboot`, `pct stop`, `pct destroy` - `pct start|shutdown|reboot|stop|destroy`
- `qm start`, `qm shutdown`, `qm reboot`, `qm stop`, `qm destroy` - `qm start|shutdown|reboot|stop|destroy`
- `docker restart`, `docker stop`, `docker rm`, `docker compose up/down` - `docker restart|stop|rm|compose up|compose down`
- package updates, firewall edits, network edits, storage edits - Package updates, firewall edits, network edits, storage edits
- Any Cloudflare write (`POST`/`PUT`/`PATCH`/`DELETE`)
- `Install-CoolifyAutostart.ps1` without `-VerifyOnly`
- `Apply-ChatwootEnterprisePatch.ps1` without `-DryRun`
+20 -8
View File
@@ -1,6 +1,6 @@
--- ---
name: coolify-agent name: coolify-agent
description: Operate the local self-hosted Coolify stack for this Proxmox and Coolify Manager project. Use when Codex needs to inspect Coolify, Docker containers inside LXC 102, applications, services, databases, deployments, logs, environment variables, or Coolify API resources on the local infrastructure. description: Operate the local self-hosted Coolify stack for this Proxmox and Coolify Manager project. Use when the agent needs to inspect Coolify, Docker containers inside LXC 102, applications, services, databases, deployments, logs, environment variables, or Coolify API resources on the local infrastructure.
--- ---
# Coolify Agent # Coolify Agent
@@ -11,14 +11,18 @@ changing production state.
## Startup Routine ## Startup Routine
1. Read `README.md`, `docs/proxmox-inventory.md`, and 1. Load secrets: `. .\.env.local.ps1`.
`docs/runbooks/coolify-docker.md`. 2. Read [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — the canonical tool
2. For app-specific work, also read the matching runbook, for example catalog. **Its §1 lists four gotchas that produce silently wrong results**;
two of them (`-Raw` inversion, `/applications` 404) bite on every Coolify task.
3. Read [`docs/proxmox-inventory.md`](../docs/proxmox-inventory.md) for current
state, and `docs/runbooks/coolify-docker.md` for the procedure.
4. For app-specific work, also read the matching runbook, for example
`docs/runbooks/nextcloud.md`. `docs/runbooks/nextcloud.md`.
3. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work. 5. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
4. Discover current state before acting: 6. Discover current state before acting:
`.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All`. `.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All`.
5. Use the Coolify API only when `COOLIFY_TOKEN` is loaded in the shell. 7. Use the Coolify API only when `COOLIFY_TOKEN` is loaded in the shell.
## Local Context ## Local Context
@@ -27,7 +31,15 @@ changing production state.
- Docker is not managed directly on Proxmox; use - Docker is not managed directly on Proxmox; use
`pct exec 102 -- docker ...` through `.\scripts\Invoke-ProxmoxSsh.ps1`. `pct exec 102 -- docker ...` through `.\scripts\Invoke-ProxmoxSsh.ps1`.
- Default Coolify API base URL: - Default Coolify API base URL:
`https://coolify.urieljareth.org/api/v1`. `https://coolify.urieljareth.org/api/v1`. Coolify runs **v4.3.14** (verified
2026-08-29) and its REST API is **complete against the origin**
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`): the
`/applications/*` and `/github-apps` 404s only happen through the public
Cloudflare hostname — an edge block, not a Coolify limitation. Call those
endpoints against the origin. State-changing endpoints are POST-only since
v4.2. Otherwise enumerate apps via `/resources`.
- **Container names are `<service>-<uuid>`** and change when a redeploy recreates
the container. Never hardcode one — resolve it first.
- Secrets must live in local environment files or the OS secret store, never in - Secrets must live in local environment files or the OS secret store, never in
Markdown or skill references. Markdown or skill references.
+79 -6
View File
@@ -2,6 +2,33 @@
All commands assume PowerShell from the project root. All commands assume PowerShell from the project root.
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md). This file
> holds usage examples; the index holds verified signatures, env requirements,
> read-only/mutating classification and the gotchas. Read its §1 first.
## Two gotchas specific to this wrapper
**1. `-Raw` is inverted.** `Invoke-CoolifyApi.ps1` returns a **JSON string** by
default and **PowerShell objects with `-Raw`**. Filtering the default output
silently yields nothing — no error:
```powershell
# WRONG: $r is a System.String, so this prints an empty row
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" | Select-Object name, uuid
# RIGHT
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw | Select-Object name, uuid
```
**2. `/applications/*` 404s through the public hostname — that block is
Cloudflare's, not Coolify's** (re-verified 2026-08-29 on v4.3.14: same token,
same route → 200 against the origin `http://192.168.0.117:8000/api/v1`, 404 via
`https://coolify.urieljareth.org`). For those endpoints use
`$env:COOLIFY_API_URL_ORIGIN`. Also: state-changing endpoints are POST-only
since v4.2 (`GET /deploy` → 405). Endpoints that work through either path:
`/version`, `/resources`, `/services`, `/databases`, `/projects`, `/servers`,
`/teams`, `/deployments`. Use `/resources` to enumerate apps.
## Environment ## Environment
Load private values from an ignored local file: Load private values from an ignored local file:
@@ -31,6 +58,29 @@ $env:COOLIFY_TOKEN = "REPLACE_WITH_TOKEN"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker volume ls" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker volume ls"
``` ```
## A New Service Is Green In Coolify But Its Domain Returns 503
Coolify's green dot means the container is `running`. Traefik only routes a
container Docker reports `healthy`. A brand-new service is `running` long before
it is `healthy`, so its route does not exist yet and the request falls through to
Coolify's catch-all (`noop` service, empty server list) — which is what prints
`no available server`.
```powershell
# Contrast running vs healthy, flag a too-short start_period, probe the domain.
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <resource-uuid>
# First boot on this host can take minutes (HDD-backed loopback rootfs). Wait it out.
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <resource-uuid> -WaitSeconds 600
```
Read the status code before changing anything:
- `502 Bad Gateway` — the route exists, the backend refuses. App or port problem.
- `503 no available server` — there is no route. Usually still starting.
**Do not redeploy**: that restarts the entrypoint from scratch and restarts the
slow boot. See TOOL-INDEX.md 1.5.
## Logs And Inspect ## Logs And Inspect
```powershell ```powershell
@@ -40,14 +90,37 @@ $env:COOLIFY_TOKEN = "REPLACE_WITH_TOKEN"
## Coolify API ## Coolify API
Verified working on this instance (2026-08-07):
```powershell ```powershell
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/version" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/version"
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" # 28 — the app inventory
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects" # 7
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services" # 14
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/databases" # 5
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/databases" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers" # 1
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/teams" # 1
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments" # in-flight deploys
```
**`/applications` and its whole namespace 404 through the public hostname
(Cloudflare edge block) but work via the origin** — point
`COOLIFY_API_URL` at `http://192.168.0.117:8000/api/v1` for those calls. Either
way, to enumerate applications and resolve a name or uuid, `/resources` is the
reliable inventory:
```powershell
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
Where-Object { $_.name -match 'chatwoot' } | Select-Object name, uuid, fqdn
```
### Resolving a container name from a resource
Container names are `<service>-<uuid>` plus an optional build suffix, so they are
not guessable and **change when a redeploy recreates the container**:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i <app>"
``` ```
For write calls, prepare the JSON body first and ask for confirmation: For write calls, prepare the JSON body first and ask for confirmation:
+6 -2
View File
@@ -47,8 +47,12 @@ if ($PSBoundParameters.ContainsKey("BodyJson")) {
throw "BodyJson is not valid JSON: $($_.Exception.Message)" throw "BodyJson is not valid JSON: $($_.Exception.Message)"
} }
$request.Body = $BodyJson # Send bytes, not a string. PowerShell 5.1 encodes a string body using the
$request.ContentType = "application/json" # default codepage, so any non-ASCII character (an accented comment inside a
# compose file, for instance) reaches Coolify mangled and it answers
# 400 {"message":"Invalid request.","error":"Invalid JSON."}.
$request.Body = [Text.Encoding]::UTF8.GetBytes($BodyJson)
$request.ContentType = "application/json; charset=utf-8"
} }
try { try {
@@ -0,0 +1,307 @@
<#
.SYNOPSIS
Give a Coolify service's healthchecks a start_period long enough for a first
boot on this host. Dry-run by default.
.DESCRIPTION
Coolify's library templates ship healthchecks tuned for SSD hosts: a short
interval, a handful of retries and no start_period at all. On this host a
first boot takes minutes (rootfs is ext4 over loopback over an HDD, ~39 ms
per write), so the container is flagged `unhealthy` long before the app
listens. Traefik only routes containers Docker reports `healthy`, so an
unhealthy container has NO route and the request falls through to Coolify's
catch-all (`noop`, empty server list) -> 503 "no available server".
Worse, once flagged unhealthy the container is a candidate for recreation,
and recreating restarts the slow entrypoint from zero. That is what turns a
transient 503 into a permanent one.
This script edits `services.docker_compose_raw` (the editable template
Coolify regenerates the deployed compose from) and inserts a `start_period`
into every healthcheck that lacks one, optionally raising a very short
`interval`.
It does NOT redeploy. The new healthcheck only takes effect when the
container is recreated, which is a separate, explicit step.
Honest scope: start_period does NOT make the site answer sooner. During
startup Docker reports `starting`, which Traefik does not route either, so
the startup 503 window still exists. What it prevents is the container being
*marked failed* and entering the recreation loop.
.PARAMETER Uuid
Coolify service uuid (last path segment of the service URL in the UI).
.PARAMETER StartPeriodSeconds
Grace window to insert. Default 300 (measured first boots here ran into the
low minutes).
.PARAMETER MinIntervalSeconds
Raise any `interval` below this. A 2 s interval spawns a health exec every
two seconds against an already saturated disk. Default 10. Pass 0 to leave
every interval untouched.
.PARAMETER Apply
Actually write. Without it the script only prints the diff and changes
nothing.
.PARAMETER ShowResult
Also print the resulting healthcheck blocks so the exact YAML can be
reviewed before writing.
.EXAMPLE
# Inspect what would change. Safe, read-only.
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid uyn0js6pqbwo8mubw5edy95f
.EXAMPLE
# Write it, then redeploy that service yourself from the Coolify UI.
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid uyn0js6pqbwo8mubw5edy95f -Apply
#>
[CmdletBinding()]
param(
[Parameter(Mandatory = $true)]
[string]$Uuid,
[int]$StartPeriodSeconds = 300,
[int]$MinIntervalSeconds = 10,
[switch]$Apply,
[switch]$ShowResult
)
$ErrorActionPreference = "Stop"
$repoRoot = Resolve-Path (Join-Path $PSScriptRoot "..\..")
$invokeSsh = Join-Path $repoRoot "scripts\Invoke-ProxmoxSsh.ps1"
$agentScript = Join-Path $repoRoot "scripts\ProxmoxAgent.ps1"
foreach ($required in @($invokeSsh, $agentScript)) {
if (-not (Test-Path -LiteralPath $required)) { throw "Missing dependency: $required" }
}
. $agentScript
$config = Get-ProxmoxConfig
$lxc = $config.CoolifyLxc
# Nested quoting is corrupted by the SSH wrapper (TOOL-INDEX.md 1.2); base64
# every remote command. Compose bodies also travel base64 so that newlines,
# quotes and Coolify's ${...} magic survive intact.
function Invoke-InLxc {
param([Parameter(Mandatory = $true)][string]$Script)
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($Script))
return @(& $invokeSsh -Command "pct exec $lxc -- bash -c 'echo $b64 | base64 -d | bash'")
}
function Invoke-CoolifyDb {
param([Parameter(Mandatory = $true)][string]$Sql)
# psql -At: unaligned, no header. Quotes are safe inside the base64 payload.
return Invoke-InLxc -Script "docker exec coolify-db psql -U coolify -At -c ""$Sql"""
}
function Get-ComposeRaw {
param([string]$ServiceUuid)
$sql = "select encode(convert_to(docker_compose_raw,'UTF8'),'base64') from services where uuid='$ServiceUuid'"
$lines = Invoke-CoolifyDb -Sql $sql
$payload = ($lines -join '').Trim()
if (-not $payload) { return $null }
return [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String($payload))
}
function Set-ComposeRaw {
param([string]$ServiceUuid, [string]$Content)
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($Content))
$sql = "update services set docker_compose_raw = convert_from(decode('$b64','base64'),'UTF8'), updated_at = now() where uuid='$ServiceUuid'"
$out = Invoke-CoolifyDb -Sql $sql
return ($out -join ' ').Trim()
}
function Get-Indent {
param([string]$Line)
if ($Line -match '^(\s*)') { return $Matches[1].Length }
return 0
}
<#
Insert start_period into every healthcheck block that lacks one, and raise a
too-short interval. Deliberately line-based: re-serialising the YAML would
reformat Coolify's magic placeholders and its `- SERVICE_URL_X` shorthand.
#>
function Update-Healthchecks {
param(
[string[]]$Lines,
[int]$StartPeriod,
[int]$MinInterval
)
$out = New-Object 'System.Collections.Generic.List[string]'
$changes = New-Object 'System.Collections.Generic.List[object]'
$i = 0
while ($i -lt $Lines.Count) {
$line = $Lines[$i]
if ($line -notmatch '^\s*healthcheck:\s*$') {
$out.Add($line)
$i++
continue
}
$hcIndent = Get-Indent -Line $line
$out.Add($line)
$hcLineNumber = $i + 1
$i++
# Collect the block: every following line indented deeper than
# `healthcheck:` itself. Blank lines inside the block are kept.
$block = New-Object 'System.Collections.Generic.List[string]'
while ($i -lt $Lines.Count) {
$candidate = $Lines[$i]
if ($candidate.Trim() -eq '') { $block.Add($candidate); $i++; continue }
if ((Get-Indent -Line $candidate) -le $hcIndent) { break }
$block.Add($candidate)
$i++
}
# Trailing blank lines belong after the block, not inside it.
while ($block.Count -gt 0 -and $block[$block.Count - 1].Trim() -eq '') {
$block.RemoveAt($block.Count - 1)
$i--
}
$childIndent = ' ' * ($hcIndent + 2)
foreach ($b in $block) {
if ($b.Trim() -ne '') { $childIndent = ' ' * (Get-Indent -Line $b); break }
}
$hasStartPeriod = @($block | Where-Object { $_ -match '^\s*start_period\s*:' }).Count -gt 0
# Raise a too-short interval.
for ($j = 0; $j -lt $block.Count; $j++) {
if ($MinInterval -le 0) { break }
if ($block[$j] -notmatch '^(\s*)interval\s*:\s*(\S+)\s*$') { continue }
$indent = $Matches[1]
$current = $Matches[2]
$seconds = $null
if ($current -match '^(\d+(?:\.\d+)?)s$') { $seconds = [double]$Matches[1] }
elseif ($current -match '^(\d+)$') { $seconds = [double]$Matches[1] }
if ($null -ne $seconds -and $seconds -lt $MinInterval) {
$block[$j] = "${indent}interval: ${MinInterval}s"
$changes.Add([pscustomobject]@{
Line = $hcLineNumber
Kind = 'interval'
From = "interval: $current"
To = "interval: ${MinInterval}s"
})
}
break
}
if (-not $hasStartPeriod) {
$block.Add("${childIndent}start_period: ${StartPeriod}s")
$changes.Add([pscustomobject]@{
Line = $hcLineNumber
Kind = 'start_period'
From = '(absent)'
To = "start_period: ${StartPeriod}s"
})
}
foreach ($b in $block) { $out.Add($b) }
}
return [pscustomobject]@{
Lines = $out.ToArray()
Changes = $changes.ToArray()
}
}
Write-Host ""
Write-Host "Healthcheck grace - service $Uuid" -ForegroundColor Cyan
Write-Host ("-" * 72)
$original = Get-ComposeRaw -ServiceUuid $Uuid
if ($null -eq $original) {
throw "No service with uuid '$Uuid' (or its docker_compose_raw is empty). Has it been deleted? Check: Invoke-CoolifyApi.ps1 -Path /services -Raw"
}
$originalLines = $original -split "`r?`n"
$result = Update-Healthchecks -Lines $originalLines -StartPeriod $StartPeriodSeconds -MinInterval $MinIntervalSeconds
$hcCount = @($originalLines | Where-Object { $_ -match '^\s*healthcheck:\s*$' }).Count
Write-Host "healthcheck blocks found: $hcCount"
if ($result.Changes.Count -eq 0) {
Write-Host "Nothing to change: every healthcheck already has a start_period and an acceptable interval." -ForegroundColor Green
return
}
Write-Host ""
Write-Host "Proposed changes:" -ForegroundColor Yellow
$result.Changes | Format-Table Line, Kind, From, To -AutoSize
$updated = ($result.Lines -join "`n")
# Guard: the edit must only ever add/modify healthcheck lines. If the line count
# moved by more than the number of inserted lines, something went wrong.
$inserted = @($result.Changes | Where-Object { $_.Kind -eq 'start_period' }).Count
$delta = $result.Lines.Count - $originalLines.Count
if ($delta -ne $inserted) {
throw "Refusing to write: line count moved by $delta but only $inserted lines should have been inserted. The block parser mis-scoped a healthcheck."
}
if ($ShowResult) {
Write-Host ""
Write-Host "Resulting healthcheck blocks:" -ForegroundColor Cyan
$lines = $result.Lines
for ($k = 0; $k -lt $lines.Count; $k++) {
if ($lines[$k] -notmatch '^\s*healthcheck:\s*$') { continue }
$indent = ($lines[$k] -replace '\S.*$', '').Length
Write-Host (" {0,4}: {1}" -f ($k + 1), $lines[$k]) -ForegroundColor DarkGray
for ($m = $k + 1; $m -lt $lines.Count; $m++) {
if ($lines[$m].Trim() -ne '' -and (($lines[$m] -replace '\S.*$', '').Length -le $indent)) { break }
$colour = if ($lines[$m] -match 'start_period|interval') { 'Green' } else { 'DarkGray' }
Write-Host (" {0,4}: {1}" -f ($m + 1), $lines[$m]) -ForegroundColor $colour
}
Write-Host ""
}
}
if (-not $Apply) {
Write-Host "DRY RUN - nothing was written. Re-run with -Apply to persist." -ForegroundColor Cyan
Write-Host "After applying you must redeploy the service for it to take effect." -ForegroundColor Cyan
return
}
$backupDir = Join-Path $repoRoot "backups"
if (-not (Test-Path -LiteralPath $backupDir)) {
New-Item -ItemType Directory -Path $backupDir | Out-Null
}
$stamp = Get-Date -Format 'yyyyMMdd-HHmmss'
$backupFile = Join-Path $backupDir "compose-raw_${Uuid}_$stamp.yml"
[IO.File]::WriteAllText($backupFile, $original, (New-Object Text.UTF8Encoding($false)))
Write-Host "Rollback copy: $backupFile" -ForegroundColor DarkGray
$status = Set-ComposeRaw -ServiceUuid $Uuid -Content $updated
Write-Host "psql: $status"
# Read back and compare, rather than trusting the UPDATE.
$verify = Get-ComposeRaw -ServiceUuid $Uuid
if ($verify -ne $updated) {
Write-Host "VERIFY FAILED - stored content does not match what was sent." -ForegroundColor Red
Write-Host "Restore with the rollback copy above before doing anything else." -ForegroundColor Red
throw "Write-back verification failed for service $Uuid."
}
Write-Host "Verified: stored docker_compose_raw matches the intended content." -ForegroundColor Green
Write-Host ""
Write-Host "NOT redeployed. The healthcheck changes only apply once the container" -ForegroundColor Yellow
Write-Host "is recreated. Redeploy the service from the Coolify UI, then confirm:" -ForegroundColor Yellow
Write-Host " .\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid $Uuid -WaitSeconds 600" -ForegroundColor Yellow
@@ -0,0 +1,270 @@
<#
.SYNOPSIS
Read-only readiness probe for a Coolify service. Explains a 503
"no available server" instead of leaving you guessing.
.DESCRIPTION
Coolify's UI reports a service as green when its container is *running*.
Traefik, however, only puts a container in the load balancer once Docker
reports it *healthy*. On this host a first boot can take minutes (HDD-backed
loopback storage, ~39 ms/write), so a brand-new service is Running but not
yet healthy — Traefik has no route for it, the request falls through to
Coolify's catch-all router (priority -1000, service `noop`, empty server
list) and Traefik answers 503 "no available server".
This script reports both signals side by side so the gap is visible, and
tells you whether you should simply wait.
Read-only: it never restarts, redeploys or mutates anything.
.PARAMETER Uuid
Coolify resource UUID (the last path segment of the service URL in the UI).
Container names carry this as a suffix and change on every redeploy, so the
real name is resolved here rather than typed by hand.
.PARAMETER Fqdn
Public URL to probe. Defaults to whatever COOLIFY_FQDN the container carries.
.PARAMETER WaitSeconds
Poll until every container is healthy, up to this many seconds. Default 0
(report once and exit). Use 600 for a first boot on this host.
.EXAMPLE
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid znpmxv2o6ggooi6qxksiagke
.EXAMPLE
# First boot of a service from the Coolify library: wait it out.
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid znpmxv2o6ggooi6qxksiagke -WaitSeconds 600
#>
[CmdletBinding()]
param(
[Parameter(Mandatory = $true)]
[string]$Uuid,
[string]$Fqdn,
[int]$WaitSeconds = 0
)
$ErrorActionPreference = "Stop"
$repoRoot = Resolve-Path (Join-Path $PSScriptRoot "..\..")
$invokeSsh = Join-Path $repoRoot "scripts\Invoke-ProxmoxSsh.ps1"
$agentScript = Join-Path $repoRoot "scripts\ProxmoxAgent.ps1"
foreach ($required in @($invokeSsh, $agentScript)) {
if (-not (Test-Path -LiteralPath $required)) {
throw "Missing dependency: $required"
}
}
. $agentScript
$config = Get-ProxmoxConfig
$lxc = $config.CoolifyLxc
# Nested quoting is corrupted by the SSH wrapper (TOOL-INDEX.md 1.2), so every
# non-trivial remote command is base64-encoded.
function Invoke-InLxc {
param([Parameter(Mandatory = $true)][string]$Script)
$bytes = [Text.Encoding]::UTF8.GetBytes($Script)
$b64 = [Convert]::ToBase64String($bytes)
return @(& $invokeSsh -Command "pct exec $lxc -- bash -c 'echo $b64 | base64 -d | bash'")
}
function Get-ServiceContainers {
param([string]$ResourceUuid)
# Container names carry the uuid as a suffix and change on every redeploy.
$lines = Invoke-InLxc -Script @"
docker ps -a --filter "label=coolify.resourceName" --format '{{.Names}}' 2>/dev/null | grep -- '$ResourceUuid' || true
docker ps -a --format '{{.Names}}' 2>/dev/null | grep -- '$ResourceUuid' || true
"@
return @($lines | Where-Object { $_ -and $_.Trim() } | ForEach-Object { $_.Trim() } | Sort-Object -Unique)
}
function Get-ContainerReport {
param([string]$Name)
$raw = Invoke-InLxc -Script @"
docker inspect '$Name' --format '{{.State.Status}}|{{if .State.Health}}{{.State.Health.Status}}|{{.State.Health.FailingStreak}}{{else}}none|0{{end}}|{{.State.ExitCode}}|{{.State.StartedAt}}|{{.RestartCount}}|{{index .Config.Labels "coolify.serviceName"}}'
docker inspect '$Name' --format 'HC|{{if .Config.Healthcheck}}{{.Config.Healthcheck.Interval}}|{{.Config.Healthcheck.Retries}}|{{.Config.Healthcheck.StartPeriod}}{{else}}absent|0|0{{end}}'
echo "ENVFQDN|`$(docker inspect '$Name' --format '{{range .Config.Env}}{{println .}}{{end}}' 2>/dev/null | sed -n 's/^COOLIFY_FQDN=//p' | head -1)"
"@
$stateLine = @($raw | Where-Object { $_ -and $_ -notmatch '^(HC|ENVFQDN)\|' })[0]
$hcLine = @($raw | Where-Object { $_ -match '^HC\|' })[0]
$envLine = @($raw | Where-Object { $_ -match '^ENVFQDN\|' })[0]
if (-not $stateLine) { return $null }
$f = $stateLine.Split('|')
$interval = $null; $retries = $null; $startPeriod = $null
if ($hcLine) {
$h = $hcLine.Split('|')
$interval = $h[1]; $retries = $h[2]; $startPeriod = $h[3]
}
$containerFqdn = $null
if ($envLine) { $containerFqdn = $envLine.Split('|', 2)[1] }
# Docker prints healthcheck durations either as raw nanoseconds or as a Go
# duration string ("2s", "1m30s"), depending on the daemon version.
$toSeconds = {
param($value)
if (-not $value) { return $null }
if ($value -match '^\d+$') { return [math]::Round([double]$value / 1e9, 1) }
$total = 0.0; $matched = $false
foreach ($m in [regex]::Matches($value, '([\d.]+)(h|ms|m|s)')) {
$n = [double]$m.Groups[1].Value
switch ($m.Groups[2].Value) {
'h' { $total += $n * 3600 }
'm' { $total += $n * 60 }
's' { $total += $n }
'ms' { $total += $n / 1000 }
}
$matched = $true
}
if ($matched) { return [math]::Round($total, 1) }
return $null
}
return [pscustomobject]@{
Name = $Name
State = $f[0]
Health = $f[1]
FailingStreak = [int]$f[2]
ExitCode = $f[3]
StartedAt = $f[4]
RestartCount = $f[5]
ServiceName = $f[6]
HealthInterval = & $toSeconds $interval
HealthRetries = $retries
HealthStartPeriod = & $toSeconds $startPeriod
Fqdn = $containerFqdn
# A container that exited 0 and has no healthcheck is a one-shot init
# step (migrations, bucket creation). It is done, not broken.
IsOneShot = ($f[0] -eq 'exited') -and ($f[3] -eq '0') -and ($f[1] -eq 'none')
RoutableByTraefik = ($f[0] -eq 'running') -and ($f[1] -in @('healthy', 'none'))
}
}
function Get-StartupWork {
param([string]$Name)
# A container stuck in its entrypoint (apt/dpkg/chown) is starting, not broken.
$lines = Invoke-InLxc -Script @"
docker top '$Name' -o pid,stat,etime,cmd 2>/dev/null | tail -n +2 || true
"@
return @($lines | Where-Object { $_ -match '\b(chown|apt|apt-get|dpkg|unzip|tar|cp)\b' })
}
Write-Host ""
Write-Host "Coolify service readiness - $Uuid" -ForegroundColor Cyan
Write-Host ("-" * 72)
$containers = @(Get-ServiceContainers -ResourceUuid $Uuid)
if (-not $containers -or $containers.Count -eq 0) {
throw "No container found carrying uuid '$Uuid' in LXC $lxc. Has the service been deployed at all?"
}
$deadline = (Get-Date).AddSeconds($WaitSeconds)
$reports = @()
while ($true) {
$reports = @($containers | ForEach-Object { Get-ContainerReport -Name $_ } | Where-Object { $_ })
$notReady = @($reports | Where-Object { -not $_.RoutableByTraefik -and -not $_.IsOneShot })
if ($notReady.Count -eq 0 -or (Get-Date) -ge $deadline) { break }
$names = ($notReady | ForEach-Object { "$($_.Name)=$($_.Health)" }) -join ', '
$left = [int]($deadline - (Get-Date)).TotalSeconds
Write-Host " waiting ($left s left): $names" -ForegroundColor DarkGray
Start-Sleep -Seconds 10
}
$reports |
Select-Object Name, State, Health, FailingStreak,
@{ n = 'Routed'; e = { if ($_.IsOneShot) { 'n/a (one-shot)' } else { $_.RoutableByTraefik } } } |
Format-Table -AutoSize
# Healthcheck tuning is the amplifier that turns "slow boot" into "stuck 503".
# A short grace window is the amplifier that turns "slow boot" into "stuck 503":
# once flagged unhealthy, the container loses its Traefik route entirely.
$graceFloorSeconds = 180
foreach ($r in $reports) {
if ($r.Health -eq 'none' -or $r.HealthStartPeriod) { continue }
if (-not $r.HealthInterval -or -not $r.HealthRetries) { continue }
$grace = $r.HealthInterval * [int]$r.HealthRetries
if ($grace -ge $graceFloorSeconds) { continue }
Write-Host " ! $($r.Name): no start_period; flagged unhealthy after ~$grace s" -ForegroundColor Yellow
Write-Host " (interval=$($r.HealthInterval)s x retries=$($r.HealthRetries)). A first boot on this host" -ForegroundColor Yellow
Write-Host " can exceed that, and an unhealthy container has no Traefik route -> 503." -ForegroundColor Yellow
}
foreach ($r in $reports) {
if ($r.RoutableByTraefik -or $r.IsOneShot) { continue }
$work = @(Get-StartupWork -Name $r.Name)
if ($work.Count -gt 0) {
Write-Host " i $($r.Name) is still running setup work in its entrypoint:" -ForegroundColor DarkCyan
$work | ForEach-Object { Write-Host " $_" -ForegroundColor DarkCyan }
Write-Host " This is slow-but-progressing, not a failure. Wait, do not redeploy." -ForegroundColor DarkCyan
}
}
$target = $Fqdn
if (-not $target) {
$withFqdn = @($reports | Where-Object { $_.Fqdn })
if ($withFqdn.Count -gt 0) { $target = $withFqdn[0].Fqdn }
}
if ($target) {
if ($target -notmatch '^https?://') { $target = "https://$target" }
Write-Host ""
Write-Host "Probing $target" -ForegroundColor Cyan
$status = $null
$body = ''
try {
$resp = Invoke-WebRequest -Uri $target -UseBasicParsing -TimeoutSec 25
$status = [int]$resp.StatusCode
$body = [string]$resp.Content
}
catch {
if ($_.Exception.Response) {
$status = [int]$_.Exception.Response.StatusCode
try {
$reader = New-Object IO.StreamReader($_.Exception.Response.GetResponseStream())
$body = $reader.ReadToEnd()
}
catch { $body = '' }
}
else {
Write-Host " transport error: $($_.Exception.Message)" -ForegroundColor Red
}
}
if ($status) { Write-Host " HTTP $status" }
if ($status -eq 503 -and $body -match 'no available server') {
Write-Host ""
Write-Host " DIAGNOSIS: Traefik has no route for this host." -ForegroundColor Yellow
Write-Host " The request fell through to Coolify's catch-all router (priority -1000," -ForegroundColor Yellow
Write-Host " service 'noop', empty server list), which is what emits this exact string." -ForegroundColor Yellow
Write-Host " Traefik's docker provider only registers containers Docker reports healthy," -ForegroundColor Yellow
Write-Host " so an unhealthy/starting container has no route at all." -ForegroundColor Yellow
Write-Host " Note: a route that exists but whose backend refuses would return 502, not 503." -ForegroundColor Yellow
Write-Host " -> Re-run with -WaitSeconds 600 before changing any configuration." -ForegroundColor Yellow
}
elseif ($status -ge 200 -and $status -lt 400) {
Write-Host " Service is reachable and routed." -ForegroundColor Green
}
}
else {
Write-Host " (no FQDN found on the containers; pass -Fqdn to probe)" -ForegroundColor DarkGray
}
Write-Host ""
$reports
+34 -15
View File
@@ -10,26 +10,45 @@ Use this skill to take a project from "local code" to "live on
(Proxmox SSH, Coolify API, Cloudflare API) plus new GitHub scaffolding into a (Proxmox SSH, Coolify API, Cloudflare API) plus new GitHub scaffolding into a
single end-to-end pipeline. single end-to-end pipeline.
## ⚠️ Instance reality check (READ FIRST — verified 2026-07 on coolify.urieljareth.org, v4.1.2) ## ⚠️ Instance reality check (READ FIRST — re-verified 2026-08-29, v4.3.14)
This instance's REST API is **partial**: the entire `/applications/*` namespace is **The `/applications/*` 404 is a Cloudflare edge block, not a Coolify limitation.**
**404** (create public/dockerfile/private-*, list, get, PATCH, `/envs`, `/logs`). Same token, same route: 404 via `https://coolify.urieljareth.org`, **200 via the
`POST /services` only takes raw compose (no git repo). Consequences: origin `http://192.168.0.117:8000/api/v1`** (`$env:COOLIFY_API_URL_ORIGIN`) —
including `/github-apps`. The instance's own `openapi.yaml` (inside the container
at `/var/www/html/openapi.yaml`) declares the full applications namespace, so the
complete API contract applies when you call the origin. Two more facts:
- You **cannot create OR configure a git-based build-from-source app via the API** here. - **State-changing endpoints are POST-only since v4.2** — `GET /deploy?uuid=` now
`New-CoolifyApplication.ps1` (POST /applications/public) will 404. answers 405 `"This endpoint has changed to a POST request."`; use
- To configure an app that **builds from a (private) git repo** — set its build pack to `-Method POST` (see notes §10.1).
Docker Compose, its compose location, its **env vars**, and its **per-service domains** — - The origin is plain HTTP inside the LAN — fine for ops from this machine; do
you MUST drive the **web UI with Playwright**. Use `scripts/coolify-ui/*.mjs` not expose it.
(needs `COOLIFY_EMAIL`/`COOLIFY_PASSWORD` in `.env.local.ps1`).
- What DOES work via API: `/resources` (inventory+status), `/projects`, `/security/keys`, Consequences:
`/services`, and **`GET /deploy?uuid=&force=true`** (trigger a deploy of any existing app),
`GET /deployments/{uuid}` (status+logs). - Creating/configuring git-based build-from-source apps **via the API should now
work by calling the origin** (`POST /applications/public`, `/dockerfile`,
`/private-deploy-key`, …). Not yet exercised end-to-end on 4.3.14 — verify on
the next deploy before retiring the UI flow.
- The **Playwright UI flow** (`scripts/coolify-ui/*.mjs`, needs
`COOLIFY_EMAIL`/`COOLIFY_PASSWORD`) and the **direct DB INSERT** path (§8 of the
notes) remain valid fallbacks.
- What works through either path: `/resources` (inventory+status), `/projects`,
`/security/keys`, `/services`, `/deploy` (POST), `/deployments/{uuid}`.
Full playbook + gotchas (UTF-8 BOM breaks Coolify's YAML parser, "Reload Compose File" is Full playbook + gotchas (UTF-8 BOM breaks Coolify's YAML parser, "Reload Compose File" is
mandatory, don't queue concurrent deploys, PowerShell `ReadAllText` for key payloads, etc.): mandatory, don't queue concurrent deploys, PowerShell `ReadAllText` for key payloads, etc.):
**[`references/coolify-4.1.2-notes.md`](references/coolify-4.1.2-notes.md)**. Re-verify the API **[`references/coolify-4.1.2-notes.md`](references/coolify-4.1.2-notes.md)** (§11 has the
surface with `GET /version` if the instance was upgraded. 2026-08-29 re-verification). Re-verify the API surface with `GET /version` if the instance
was upgraded.
**NEW (2026-07-27):** There is also a **third path** for creating apps — **direct DB
INSERT via SSH → pct → docker exec → psql**. See
[`references/coolify-4.1.2-notes.md` §8](references/coolify-4.1.2-notes.md).
This is the fastest path when you have SSH access to the Proxmox host. The key gotcha: you
MUST also INSERT a matching `application_settings` row or deploys crash with
"disable_build_cache on null" — and stuck deploy queues must be cleared manually (§8.4).
## When to use ## When to use
+64 -15
View File
@@ -6,19 +6,51 @@ All commands assume PowerShell from the project root. Load env first:
. .\.env.local.ps1 . .\.env.local.ps1
``` ```
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — verified
> signatures, env requirements, and which scripts are broken on this instance.
## Which pipelines actually work here (re-verified 2026-08-29, v4.3.14)
Through the **public hostname**, `/applications/*` still 404s — but that block is
**Cloudflare's edge, not Coolify's**: against the origin
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`) the full
applications namespace responds 200 with the same token. State-changing
endpoints are POST-only since v4.2. Script status today:
| Script | Status | Why |
|---|---|---|
| `Publish-ProjectToCoolify.ps1` | ⚠️ untested on 4.3.14 | calls `New-CoolifyApplication.ps1` → point `COOLIFY_API_URL` at the origin |
| `New-CoolifyApplication.ps1` | ⚠️ untested on 4.3.14 | `POST /applications/public`, `PATCH /applications/{uuid}` — respond via the origin |
| `Invoke-CoolifyRollback.ps1` | ⚠️ untested on 4.3.14 | `PATCH /applications/{uuid}` — responds via the origin |
| `New-CoolifyService.ps1` | ✅ works | `POST /services` |
| `coolify-ui/*.mjs` (Playwright) | ✅ works | drives the web UI |
| `New-CoolifyAppViaDB.ps1` | ⚠️ last resort | direct `INSERT` into Coolify's DB |
**Working paths, in order of preference:**
1. Multi-container stack with `docker-compose.coolify.yml` → `New-CoolifyService.ps1`.
2. Git build-from-source app → try the API against the origin
(`New-CoolifyApplication.ps1` with `COOLIFY_API_URL` pointed at the origin);
if it misbehaves, fall back to the Playwright UI flow below.
3. Last resort → `New-CoolifyAppViaDB.ps1` (no validation, no rollback).
The scaffold/validate/verify scripts (`Initialize-CoolifyProject.ps1`,
`Test-PreDeployChecklist.ps1`, `Test-ServiceOnline.ps1`, `New-GitHubRepo.ps1`)
are unaffected and work normally.
Required env vars (in `.env.local.ps1`, gitignored): Required env vars (in `.env.local.ps1`, gitignored):
- `COOLIFY_TOKEN` — for Coolify API. - `COOLIFY_TOKEN` — for Coolify API.
- `GITHUB_TOKEN` — PAT for GitHub API (create repo, list). `git push` uses - `GITHUB_TOKEN` — PAT for GitHub API (create repo, list). `git push` uses
wincred, NOT this PAT. wincred, NOT this PAT.
- (Optional) `GITHUB_OWNER` — override the default `urieljarethbusiness-cpu`. - (Optional) `GITHUB_OWNER` — override the default `urieljarethbusiness-cpu`.
- `COOLIFY_EMAIL` / `COOLIFY_PASSWORD` — Coolify **web UI** login. Required for the - `COOLIFY_EMAIL` / `COOLIFY_PASSWORD` — Coolify **web UI** login. Required for
UI-driven flow below (this instance's `/applications/*` API is 404 — see the UI-driven fallback flow below (see `references/coolify-4.1.2-notes.md`;
`references/coolify-4.1.2-notes.md`). the probable values are already in `.env.local.ps1`).
## UI-driven config for git build-from-source apps (this instance, v4.1.2) ## UI-driven config for git build-from-source apps (fallback path)
Because `/applications/*` is 404 here, configure git-based Docker-Compose apps through the If the API-via-origin path misbehaves, configure git-based Docker-Compose apps through the
UI with Playwright (run from the manager repo root so `require('playwright')` resolves): UI with Playwright (run from the manager repo root so `require('playwright')` resolves):
```powershell ```powershell
@@ -38,8 +70,8 @@ $env:CF_ENVBULK = (Get-Content .\myapp\.env.coolify -Raw) # KEY=VALUE lines ->
$env:CF_DOMAINS = '{"web":"https://myapp.urieljareth.org"}' # pin per-service domains (else Coolify assigns random ones -> 503) $env:CF_DOMAINS = '{"web":"https://myapp.urieljareth.org"}' # pin per-service domains (else Coolify assigns random ones -> 503)
node .\deploy_skill\scripts\coolify-ui\Configure-CoolifyComposeApp.mjs node .\deploy_skill\scripts\coolify-ui\Configure-CoolifyComposeApp.mjs
# 3. Trigger the deploy (API works even though app CRUD is 404) and watch it # 3. Trigger the deploy (POST since v4.2) and watch it
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=<app>" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy?uuid=<app>"
# poll GET /deployments/{deployment_uuid} for status+logs; then verify: # poll GET /deployments/{deployment_uuid} for status+logs; then verify:
.\deploy_skill\scripts\Test-ServiceOnline.ps1 -Fqdn https://myapp.urieljareth.org -ExpectTitle "<regex>" .\deploy_skill\scripts\Test-ServiceOnline.ps1 -Fqdn https://myapp.urieljareth.org -ExpectTitle "<regex>"
``` ```
@@ -55,8 +87,16 @@ node .\deploy_skill\scripts\coolify-ui\Configure-CoolifyComposeApp.mjs
> Dockerfile builds (POST /applications/public). `New-CoolifyService.ps1` is > Dockerfile builds (POST /applications/public). `New-CoolifyService.ps1` is
> for multi-container Docker Compose stacks (POST /services). Use the right > for multi-container Docker Compose stacks (POST /services). Use the right
> one for your project shape. > one for your project shape.
>
> Since 2026-08-29 the first pipeline should also work by pointing
> `COOLIFY_API_URL` at the origin (`http://192.168.0.117:8000/api/v1`) — the 404
> on `/applications/*` was Cloudflare's edge, not Coolify's. Verify on the next
> deploy before relying on it.
### Full pipeline (one shot) ### Full pipeline (one shot) — via the origin
Blocked only through the public hostname (Cloudflare 404 on
`/applications/*`). Point the API at the origin and step 5 should succeed:
```powershell ```powershell
.\deploy_skill\scripts\Publish-ProjectToCoolify.ps1 ` .\deploy_skill\scripts\Publish-ProjectToCoolify.ps1 `
@@ -168,18 +208,27 @@ You will be prompted for the SHA if you omit `-CommitSha`.
## Coolify API (already documented in coolify_skill) ## Coolify API (already documented in coolify_skill)
The deploy skill reuses `coolify_skill/scripts/Invoke-CoolifyApi.ps1`. Useful The deploy skill reuses `coolify_skill/scripts/Invoke-CoolifyApi.ps1`. Endpoints
endpoints for the deploy flow: that **work here**, useful for the deploy flow:
```powershell ```powershell
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects"
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers"
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services"
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications/<uuid>" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" # enumerate apps + get uuids
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications/<uuid>/deployments" .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy?uuid=<uuid>" # trigger a deploy (POST-only since v4.2)
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy" -BodyJson (@{ uuid = "<uuid>" } | ConvertTo-Json) .\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments/<deployment-uuid>"
``` ```
⚠️ `/applications*` and `/github-apps` **404 through the public hostname
(Cloudflare edge block, re-verified 2026-08-29)** — call them via the origin
(`$env:COOLIFY_API_URL_ORIGIN` = `http://192.168.0.117:8000/api/v1`), where the
full namespace works. `/resources` remains the simplest inventory;
`/deployments/<uuid>` covers deployment status.
Remember `-Raw` if you need objects instead of a JSON string — see
[`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) §1.1.
For per-endpoint schemas, search `coolify_skill/references/ops/` with `rg`. For per-endpoint schemas, search `coolify_skill/references/ops/` with `rg`.
## Templates ## Templates
@@ -239,7 +288,7 @@ the manager repo root.
| Symptom | Likely cause | Fix | | Symptom | Likely cause | Fix |
|---------|--------------|-----| |---------|--------------|-----|
| `git push` asks for username/password | wincred cache miss | Run any git HTTPS op once interactively; or `cmdkey /generic:git:https://github.com /user:urieljarethbusiness-cpu /pass:<PAT>` | | `git push` asks for username/password | wincred cache miss | Run any git HTTPS op once interactively; or `cmdkey /generic:git:https://github.com /user:urieljarethbusiness-cpu /pass:<PAT>` |
| GitHub API 401 | `GITHUB_TOKEN` not loaded or expired | `. .\.env.local.ps1`; regenerate PAT at https://github.com/settings/tokens | | GitHub API 401 | `GITHUB_TOKEN` not loaded or expired | `. .\.env.local.ps1` (carries the wincred token since 2026-08-29); re-extract with `git credential fill` if it rotates |
| Coolify API 401 | `COOLIFY_TOKEN` not loaded | `. .\.env.local.ps1` | | Coolify API 401 | `COOLIFY_TOKEN` not loaded | `. .\.env.local.ps1` |
| App at 502 after deploy | `ports_exposes` mismatch (default is 3000) | `New-CoolifyApplication.ps1 -PortsExposes <real-port>` | | App at 502 after deploy | `ports_exposes` mismatch (default is 3000) | `New-CoolifyApplication.ps1 -PortsExposes <real-port>` |
| App can't reach DB | used `localhost` or wrong network | Use service name + external `coolify` network (see §2.1, §2.2 in AGENTS-coolify-apps.md) | | App can't reach DB | used `localhost` or wrong network | Use service name + external `coolify` network (see §2.1, §2.2 in AGENTS-coolify-apps.md) |
+510 -3
View File
@@ -7,6 +7,12 @@
## 1. The REST API surface is PARTIAL — `/applications/*` is 404 ## 1. The REST API surface is PARTIAL — `/applications/*` is 404
> **CORRECCIÓN (2026-08-29, §11):** ese 404 lo imponía **Cloudflare en el
> hostname público, no Coolify**. Contra el origen
> (`http://192.168.0.117:8000/api/v1`) el namespace completo responde 200 y el
> `openapi.yaml` del contenedor declara toda la superficie. Esta sección y la
> tabla se conservan como registro histórico del diagnóstico de 2026-08-07.
With a valid **root-team** token (`GET /teams/current` → "Root Team"): With a valid **root-team** token (`GET /teams/current` → "Root Team"):
| Endpoint | Result | | Endpoint | Result |
@@ -114,7 +120,7 @@ Realtime, Storage, Kong, Studio) end-to-end. New, reusable facts:
### 7.2 Locally-built images → Coolify's deploy `pull`s and fails ### 7.2 Locally-built images → Coolify's deploy `pull`s and fails
- Coolify's service deploy runs `docker compose pull` → `pull access denied ... repository does not exist` for a local-only image tag. **Don't use Coolify's Deploy button/`/start` for local images.** - Coolify's service deploy runs `docker compose pull` → `pull access denied ... repository does not exist` for a local-only image tag. **Don't use Coolify's Deploy button/`/start` for local images.**
- Instead build the image on the server (clone repo + `docker build -t <tag>`), then `docker compose up -d` **manually** in `/data/coolify/services/<uuid>/` (default pull policy skips pull when the image exists locally). Same pattern as `Deploy-SoloLeveling.ps1`. - Instead build the image on the server (clone repo + `docker build -t <tag>`), then `docker compose up -d` **manually** in `/data/coolify/services/<uuid>/` (default pull policy skips pull when the image exists locally). Same pattern as `scripts/apps/Deploy-SoloLeveling.ps1`.
- Coolify's normalized on-disk compose (from your base64 raw) **preserves** your `image`, inlined `environment`, custom `labels` (incl. Traefik) and networks — but **renames containers to `<service>-<uuid>`**. Cross-container refs must use the *other* service's real name (unchanged), not the renamed one. - Coolify's normalized on-disk compose (from your base64 raw) **preserves** your `image`, inlined `environment`, custom `labels` (incl. Traefik) and networks — but **renames containers to `<service>-<uuid>`**. Cross-container refs must use the *other* service's real name (unchanged), not the renamed one.
- The service dir + its per-service external network `<uuid>` are created only on Coolify's own deploy. For a first manual `up`, run `docker network create --attachable <uuid>` first (else `network <uuid> declared as external, but could not be found`). Put app containers on the external `coolify` network to reach other stacks (Traefik `coolify-proxy` is already on it). - The service dir + its per-service external network `<uuid>` are created only on Coolify's own deploy. For a first manual `up`, run `docker network create --attachable <uuid>` first (else `network <uuid> declared as external, but could not be found`). Put app containers on the external `coolify` network to reach other stacks (Traefik `coolify-proxy` is already on it).
@@ -141,7 +147,7 @@ Every deploy script here sets `$ErrorActionPreference = "Stop"` and shells out t
kills the script on its first line of benign progress (e.g. git `Cloning into 'repo'...`). kills the script on its first line of benign progress (e.g. git `Cloning into 'repo'...`).
Symptom: exit 1 with the error anchored at `& ssh @sshArgs` in `ProxmoxAgent.ps1`, right after the Symptom: exit 1 with the error anchored at `& ssh @sshArgs` in `ProxmoxAgent.ps1`, right after the
first `git`/`ssh` progress line — **before any real work fails**. first `git`/`ssh` progress line — **before any real work fails**.
- **Fix:** invoke the script **plain** — `& .\Deploy-SoloLeveling.ps1` (no `2>&1`/`*>&1`, no - **Fix:** invoke the script **plain** — `& .\scripts\apps\Deploy-SoloLeveling.ps1` (no `2>&1`/`*>&1`, no
merging pipe). The harness/terminal already captures the process's stderr at the OS level, which merging pipe). The harness/terminal already captures the process's stderr at the OS level, which
does **not** create ErrorRecords. Same rule for `Publish-ProjectToCoolify.ps1`, does **not** create ErrorRecords. Same rule for `Publish-ProjectToCoolify.ps1`,
`New-CoolifyService.ps1`, `Invoke-CoolifyRollback.ps1`, `Test-ServiceOnline.ps1`. `New-CoolifyService.ps1`, `Invoke-CoolifyRollback.ps1`, `Test-ServiceOnline.ps1`.
@@ -150,7 +156,7 @@ first `git`/`ssh` progress line — **before any real work fails**.
- Note the scripts' *internal* `git push 2>&1 | ForEach-Object {…}` is fine (scoped to one - Note the scripts' *internal* `git push 2>&1 | ForEach-Object {…}` is fine (scoped to one
statement); the trap is the **outer** redirect the caller adds. statement); the trap is the **outer** redirect the caller adds.
### 7.8 build-on-server re-deploy: verification + rollback caveats (Solo Leveling, `Deploy-SoloLeveling.ps1`) ### 7.8 build-on-server re-deploy: verification + rollback caveats (Solo Leveling, `scripts/apps/Deploy-SoloLeveling.ps1`)
- **Confirm the shipped commit, not just health.** The build step echoes `HEAD: <sha> <subject>` - **Confirm the shipped commit, not just health.** The build step echoes `HEAD: <sha> <subject>`
from the fresh clone — gate on it matching your intended commit, and confirm the app container from the fresh clone — gate on it matching your intended commit, and confirm the app container
shows **`Recreated`** (not reused) in `docker compose up -d` output. Health 200 alone only proves shows **`Recreated`** (not reused) in `docker compose up -d` output. Health 200 alone only proves
@@ -162,3 +168,504 @@ first `git`/`ssh` progress line — **before any real work fails**.
If instant rollback matters, tag per-commit too (`:<sha>` alongside `:latest`) so you can retag If instant rollback matters, tag per-commit too (`:<sha>` alongside `:latest`) so you can retag
`:latest` to a prior digest and `docker compose up -d`. (The skill's `Invoke-CoolifyRollback.ps1` `:latest` to a prior digest and `docker compose up -d`. (The skill's `Invoke-CoolifyRollback.ps1`
assumes the git-build path, which is `/applications/*` = 404 here — it doesn't apply to build-on-server.) assumes the git-build path, which is `/applications/*` = 404 here — it doesn't apply to build-on-server.)
## 8. Creating git-based apps via direct DB INSERT (when UI credentials are unavailable)
> Verified 2026-07-27 deploying `AgendaMax` (Node 22 + Vite + Express + SQLite,
> Dockerfile build pack, GitHub source). This is the **third** path, alongside
> §3 (Playwright UI) and §6 (API, 404 here). Use it when you have SSH+DB access
> but NOT the Coolify web UI email/password.
### 8.1 The SSH → pct → docker exec → psql chain
```
local machine
→ ssh [email protected] -i keys\proxmox_ed25519
→ pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -c "SQL_HERE"
```
**Credentials (all local/private, no secrets leave the host):**
- Proxmox host: `[email protected]`
- SSH key: `keys\proxmox_ed25519`
- Coolify LXC: `102`
- Coolify DB container: `coolify-db` (PostgreSQL 15, user `coolify`, db `coolify`, no password — container-internal)
- Coolify app container: `coolify` (image `ghcr.io/coollabsio/coolify:4.1.2`)
- Coolify API token: in `.env.local.ps1` → `$env:COOLIFY_TOKEN`
- GitHub PAT: in `.env.local.ps1` → `$env:GITHUB_TOKEN` (scope `repo`, account `urieljarethbusiness-cpu`)
### 8.2 PowerShell quoting nightmare — the scp+sh workaround
**Problem:** Passing SQL (with single quotes, backslashes in PHP namespaces like
`App\Models\GithubApp`, or `{{pr_id}}` braces) through the chain
PowerShell → ssh → pct → docker → psql is quoting hell. Nested `'…'` inside
`"…"` inside `"…"` breaks at every level.
**Solution:** Write a `.sh` script locally → `scp` to the Proxmox host → `ssh … sh /tmp/script.sh`.
```powershell
# 1. Write the script locally with heredoc-safe content
# 2. scp it to the host
scp -i $SSH_KEY script.sh root@192.168.0.200:/tmp/script.sh
# 3. Execute remotely
ssh -i $SSH_KEY root@192.168.0.200 "sh /tmp/script.sh"
```
**Never** try to pass complex SQL inline through `ssh … "pct exec … psql … -c '…'"` from
PowerShell — the nested quoting will eat hours. Always scp a script.
### 8.3 Step-by-step: create a Dockerfile-based app from a GitHub repo
**Prerequisites:**
- The repo must be on GitHub (`urieljarethbusiness-cpu/<name>`) — Coolify's GitHub App
(source_id=3, installation_id=121211999) is already configured and can access all repos
under that account.
- Push the code first: `git push github main`.
- Get the GitHub repo numeric ID: `GET https://api.github.com/repos/<owner>/<repo>` → `.id`.
**Step 1 — Find the environment_id:**
```sql
SELECT id FROM environments WHERE uuid = '<env_uuid>';
-- e.g. for project "tools" production: returns 9
```
**Step 2 — INSERT into `applications`:** every field with a NOT NULL DEFAULT in the schema
must be set explicitly (the INSERT doesn't fire Laravel's model events, so defaults from
migrations are the DB column defaults, not Eloquent `$casts`/boot logic).
Key fields for a Dockerfile app:
```sql
INSERT INTO applications (
uuid, name, git_repository, git_branch, git_commit_sha,
build_pack, -- 'dockerfile'
dockerfile_location, -- '/Dockerfile'
ports_exposes, -- '3000' (your app's port)
health_check_path, health_check_port, health_check_host,
health_check_method, health_check_return_code, health_check_scheme,
health_check_interval, health_check_timeout, health_check_retries, health_check_start_period,
health_check_enabled, -- false (simpler; enable later via UI if needed)
limits_memory, limits_memory_swap, limits_memory_swappiness, limits_memory_reservation,
limits_cpus, limits_cpu_shares,
status, -- 'exited'
preview_url_template, -- '{{pr_id}}.{{domain}}'
fqdn, -- 'https://<name>.urieljareth.org'
repository_project_id, -- numeric GitHub repo ID (from GitHub API)
source_type, -- 'App\Models\GithubApp'
source_id, -- 3 (the GitHub App, see SELECT id FROM github_apps)
destination_type, -- 'App\Models\StandaloneDocker'
destination_id, -- 0 (the localhost Docker, see SELECT id FROM standalone_dockers)
environment_id, -- from Step 1
base_directory, -- '/'
static_image, -- 'nginx:alpine' (unused for dockerfile pack, but NOT NULL)
created_at, updated_at -- NOW()
) VALUES (...);
```
**Step 3 — INSERT into `application_settings` (CRITICAL — skip this and deploys crash):**
```
production.ERROR: Attempt to read property "disable_build_cache" on null
at ApplicationDeploymentJob.php:214
```
Coolify's deployment job reads `$application->settings->disable_build_cache` in its
constructor. Without a settings row, `settings` is null and the job dies instantly —
the deployment stays "in_progress" forever and blocks the queue (see §8.4).
```sql
INSERT INTO application_settings (
application_id, -- the id from Step 2's RETURNING
is_static, is_git_submodules_enabled, is_git_lfs_enabled,
is_auto_deploy_enabled, is_force_https_enabled, is_debug_enabled,
is_preview_deployments_enabled, is_log_drain_enabled, is_gpu_enabled,
is_swarm_only_worker_nodes, is_raw_compose_deployment_enabled,
is_build_server_enabled, is_consistent_container_name_enabled,
is_gzip_enabled, is_stripprefix_enabled,
is_container_label_escape_enabled, is_container_label_readonly_enabled,
disable_build_cache, is_spa, is_git_shallow_clone_enabled,
is_pr_deployments_public_enabled, use_build_secrets, inject_build_args_to_dockerfile,
docker_images_to_keep,
created_at, updated_at
) VALUES (
<app_id>, false, true, true, true, true, false, false, false, false,
true, false, false, false, true, true, true, true, false, false, true,
false, false, true, 2, NOW(), NOW()
);
```
**Step 4 — Trigger deploy via API (works even though app CRUD is 404):**
```powershell
. .\.env.local.ps1
$h = @{ Authorization = "Bearer $env:COOLIFY_TOKEN" }
Invoke-RestMethod "$env:COOLIFY_API_URL/deploy?uuid=<app_uuid>&force=true" -Headers $h
# Returns: { deployments: [{ deployment_uuid: "..." }] }
```
Monitor: `GET /deployments/<deployment_uuid>` → status goes `queued → in_progress → finished`.
### 8.4 Stuck deployment queue — how to unblock
When a deployment fails (e.g. missing `application_settings`), the queue row stays
`in_progress` forever and blocks ALL subsequent deploys of that app.
**Diagnose:**
```sql
SELECT id, status, created_at FROM application_deployment_queues
WHERE application_id = '<app_id>' ORDER BY id DESC LIMIT 5;
```
**Fix:**
```sql
-- Mark the stuck job as failed
UPDATE application_deployment_queues SET status = 'failed', updated_at = NOW()
WHERE id = <stuck_id>;
-- Delete ALL queue entries for the app and start clean
DELETE FROM application_deployment_queues WHERE application_id = '<app_id>';
```
Then restart the Coolify queue worker (so it picks up new jobs):
```sh
pct exec 102 -- docker exec coolify php artisan queue:restart
```
Wait 5 seconds, then trigger a fresh deploy via the API.
### 8.5 Reference: existing GitHub-source apps on this instance (verified 2026-07-27)
| id | name | repo | build_pack | source_id |
|----|------|------|------------|-----------|
| 43 | baserow | baserow/baserow | dockercompose | 3 |
| 46 | estación-de-documentos | urieljarethbusiness-cpu/Estación-de-Documentos | dockercompose | 3 |
| 47 | cotizador | urieljarethbusiness-cpu/cotizador | dockercompose | 3 |
| 48 | firecrawl | firecrawl/firecrawl | dockercompose | 3 |
| 50 | audio-a-texto | urieljarethbusiness-cpu/Audio-a-Texto | dockerfile | 3 |
| 51 | agendamax | urieljarethbusiness-cpu/agendamax | dockerfile | 3 |
GitHub App: id=3, uuid=`miakw8c0vthrtweroh9kzy1t`, app_id=3266541, installation_id=121211999.
Public GitHub source: id=0, uuid=`yyq0od5j3xkdf8twc28n6coh`.
Standalone Docker (destination): id=0, uuid=`jxlxd62k0d8owl6orgkjj1ob`, network=`coolify`.
Server: id=0, uuid=`l10mdaago0z605pga93gl6cz`, name=`localhost`.
### 8.6 Gitea repos as build source — not directly supported via this path
Coolify's GitHub App source only works with GitHub.com repos. For a Gitea-hosted repo:
- **Option A (recommended):** mirror to GitHub (`git remote add github …; git push github main`),
then create the Coolify app from the GitHub repo. This is what we did for AgendaMax —
the primary repo stays on Gitea, GitHub is just a deploy mirror.
- **Option B:** create a "Private repository (deploy key)" app via the web UI — set
`git_repository` to the Gitea SSH URL and `private_key_id` to an SSH key with Gitea access.
Gitea SSH is on port `22222` (mapped from container 22). This path needs UI credentials
and hasn't been tested on this instance.
Re-verified 2026-07-31 (`prompt-gallery-e3`): Option A still works, and the GitHub App
(source_id=3) reaches a **freshly created** private repo under `urieljarethbusiness-cpu`
with no extra configuration — the installation covers all repos on the account, so there
is nothing to click between `POST /user/repos` and the first deploy.
### 8.7 Env vars are Laravel-`encrypted` — raw SQL INSERT kills the deploy
> Verified 2026-07-31 deploying `prompt-gallery-e3` (Next.js 16 + `node:sqlite`).
> §8.3 creates the app but says nothing about env vars; this is the trap.
`environment_variables.value` carries Laravel's `encrypted` cast. Insert a **plaintext**
value with `psql` and the app row looks perfect, but every deploy dies **after** cloning
and reading the Dockerfile, with no hint about which field is at fault:
```
Deployment failed: The payload is invalid.
Error type: Illuminate\Contracts\Encryption\DecryptException
Location: /var/www/html/vendor/laravel/framework/src/Illuminate/Encryption/Encrypter.php:244
```
A correctly stored value is base64 JSON (`eyJpdiI6…`, `{iv,value,mac,tag}`), 200–300 chars
for short secrets. The `APP_KEY` never leaves the container, so **don't try to encrypt from
the outside** — create the rows through Eloquent instead:
```powershell
# write PHP locally -> scp to host -> pct push -> docker cp -> tinker
pct push 102 /tmp/envs.php /tmp/envs.php
pct exec 102 -- docker cp /tmp/envs.php coolify:/tmp/envs.php
pct exec 102 -- docker exec coolify php artisan tinker --execute='include "/tmp/envs.php";'
```
```php
$fila = new \App\Models\EnvironmentVariable();
$fila->key = 'SESSION_SECRET';
$fila->value = '…'; // se cifra al guardar
$fila->resourceable_type = \App\Models\Application::class;
$fila->resourceable_id = 53;
$fila->save(); // uuid se autogenera
```
Three gotchas found doing this:
- **DELETE then CREATE — never UPDATE a plaintext row.** Saving over an existing bad row
throws the same `DecryptException` first, because a model hook reads `value` before your
assignment is written. Wipe the rows with SQL, then create them via Eloquent.
- **The column is `is_buildtime`, not `is_build_time`** (and `is_runtime`/`is_buildtime`
are set by the model — don't touch them).
- **Two rows per key is correct.** Coolify mirrors every variable into a preview copy
(`is_preview = true`), so ids come in pairs. The pre-existing apps show the same shape;
it is not a duplicate-insert bug.
### 8.8 Persistent volumes by this path
`local_persistent_volumes` uses `resource_type` / `resource_id` — **not** the
`resourceable_*` names that `environment_variables` uses. Plain SQL is fine here (no
encrypted columns):
```sql
INSERT INTO local_persistent_volumes (name, mount_path, resource_type, resource_id, uuid, created_at, updated_at)
VALUES ('<app_uuid>-galeria-datos', '/app/data', 'App\Models\Application', <app_id>, '<uuid24>', NOW(), NOW());
```
Insert the volume **before** the first deploy. `New-CoolifyAppViaDB.ps1` triggers a deploy
as its last step, so for an app that needs env vars or a volume, do the whole set of rows
in one data-modifying-CTE statement and trigger `GET /deploy?uuid=` yourself — otherwise
the first build is guaranteed to fail and you burn ~15 min of Next.js build time.
### 8.9 Script bugs fixed 2026-07-31 (were silently breaking repo creation)
- **`" 20[04-9] "` accepted 200 and 204–209 but rejected 201/202/203** — a character-class
typo for `[0-9]`. Present in **both** `gitea_skill/scripts/Invoke-GiteaApi.ps1` and
`deploy_skill/scripts/Invoke-GitHubApi.ps1`. Every repo/hook/release creation returns
**201**, so it threw *after* successfully creating the resource — leaving a created repo
and an aborted pipeline. Both fixed to `" 20[0-9] "`.
- **`Invoke-GitHubApi.ps1` BOM bug (documented in §7.3) is now actually fixed** — the POST
body is written with `[IO.File]::WriteAllText(..., UTF8Encoding($false))`.
- **`git push` to GitHub: `Authorization: Bearer <PAT>` does NOT work.** git-over-https
wants Basic. For a headless push that leaves no token on disk:
```powershell
$basic = [Convert]::ToBase64String([Text.Encoding]::ASCII.GetBytes("x-access-token:$env:GITHUB_TOKEN"))
git -C $repo -c "http.extraHeader=Authorization: Basic $basic" -c "credential.helper=" push -u github main
```
(`Sync-GiteaRemote.ps1`'s `token <T>` header is right for Gitea; it is not for GitHub.)
- `New-GitHubRepo.ps1` forces `auto_init = $true`, which puts a commit on the remote and
makes pushing existing history a non-fast-forward. For a mirror of a repo that already
has history, POST `/user/repos` yourself with `auto_init = $false`.
## 9. Apps with a sibling database (verified 2026-07-31, `demospa` — Next.js 15 + Prisma + MySQL 8)
Deployed `serenidad-spa` as `demospa.urieljareth.org` (app id 54) with a Coolify-managed
MySQL. New, reusable facts beyond §8:
### 9.1 `POST /databases/mysql` WORKS — the database API is not part of the 404 namespace
Unlike `/applications/*`, the `/databases/*` namespace responds. `POST /databases/mysql`
with `{server_uuid, project_uuid, environment_name, environment_uuid, destination_uuid,
name, image, mysql_root_password, mysql_database, mysql_user, mysql_password}` returns
`{uuid, internal_db_url}`. **The DB's internal hostname IS its uuid** — so
`DATABASE_URL=mysql://user:pass@<db-uuid>:3306/<db>`. Both the DB and dockerfile-pack apps
land on the external `coolify` network, so service-name DNS works with no extra wiring.
Two gotchas:
- **`instant_deploy: true` did NOT start it.** The resource was created with
`status=exited:unhealthy` and no container. `GET /databases/{uuid}/start` then answered
`400 {"message":"Database is already running."}` (the status field lies). What actually
started it: **`GET /deploy?uuid=<db-uuid>`** — the same trigger used for apps.
- **Never call `/start` and `/deploy` back to back.** Doing so recreated the container
mid-initialization and left a partial datadir; MySQL then crash-looped forever on
`--initialize specified but the data directory has files in it`. Recovery =
`docker compose down` in `/data/coolify/databases/<uuid>/`, `docker volume rm
mysql-data-<uuid>`, then one clean `up -d`. Same "don't queue deploys" rule as §4.
### 9.2 MySQL 8 init on this host takes ~7 min, and `start_period` is hardcoded to 5s
Coolify's generated DB compose sets `healthcheck.start_period: 5s`, but a first-time
MySQL 8 init here takes minutes (InnoDB init alone ~53s). Patch
`/data/coolify/databases/<uuid>/docker-compose.yml` to a longer `start_period` before the
first `up -d`.
**Do not trust `mysqladmin ping` as a readiness signal.** The MySQL entrypoint runs a
*temporary* server with `--skip-networking` while it creates the database and user, so the
socket answers (and Coolify reports `healthy`) while TCP still refuses connections. The log
even prints `ready for connections ... port: 0` for that temp server. Gate on TCP:
`mysqladmin -h127.0.0.1 -uroot -p"$MYSQL_ROOT_PASSWORD" ping`, and confirm a
`ready for connections ... port: 3306` line.
### 9.3 Prisma + MySQL 8: pin `mysql_native_password`
Add to the DB service's compose `command:`
`--default-authentication-plugin=mysql_native_password --character-set-server=utf8mb4
--collation-server=utf8mb4_unicode_ci`. 8.0.46 only warns that the flag is deprecated. Set
it **before** the first init so the created user gets native auth (afterwards the user
already exists and the env vars are ignored — you'd need `ALTER USER`).
### 9.4 The rolling-update healthcheck window is ~2 minutes — do slow work in the background
Coolify polls the container's **Dockerfile `HEALTHCHECK`** ~6 times at 30s and aborts with
"New container is not healthy, rolling back" if it hasn't passed. An entrypoint that runs
`prisma db push` + seed *before* starting the server will lose this race on a cold DB — the
build succeeds and the deploy still fails.
Working shape: start the server in the foreground and run DB preparation in a background
subshell, so the health endpoint answers in seconds while data lands moments later.
```sh
preparar_base_de_datos() { ...db push retry loop...; ...seed...; }
preparar_base_de_datos &
exec "$@" # el proceso en segundo plano sobrevive al exec
```
Keep the Dockerfile healthcheck tight (`--start-period=10s --interval=10s`) so it turns
healthy inside Coolify's window, and give the health endpoint **no DB dependency**.
### 9.5 `inject_build_args_to_dockerfile` bakes every env var into the image
`New-CoolifyAppViaDB.ps1`'s settings row sets this `true`, so Coolify rewrites the
Dockerfile with an `ARG`/`ENV` per variable — the build log fills with
`SecretsUsedInArgOrEnv: ... (ARG "AUTH_SECRET")` and the secrets end up in image layers,
violating §2.5 of `AGENTS-coolify-apps.md`. Set it to `false` unless the app genuinely needs
build-time vars (a Next.js app only does if it reads `NEXT_PUBLIC_*`, which are inlined at
build time — grep the source before deciding).
### 9.6 Pushing to the GitHub mirror auto-triggers a deploy
The settings row also sets `is_auto_deploy_enabled = true`, and Coolify's GitHub App gets
push webhooks for the whole account — so `git push github main` queues a deploy on its own.
Expect an extra `in_progress` row in `application_deployment_queues`; clear it (§8.4) before
triggering your own, or just let the automatic one run.
### 9.7 Next.js: pages that query the DB break `docker build`
Any App Router page that hits the database without `export const dynamic = "force-dynamic"`
is prerendered during `next build`, where no DB exists. Pages reading `cookies()` are
already dynamic; public landing/catalog pages usually are not. Validate locally with a
deliberately unreachable `DATABASE_URL` — the build must still finish (Prisma logs errors
but they are non-fatal once every DB page is `ƒ`).
`sharp` needs no special handling: `npm ci` on `node:22-alpine` resolves
`@img/sharp-linuxmusl-x64` as long as the lockfile was generated with all platform variants
(it is, by default), so remote-image optimization works.
---
## 10. Correcciones verificadas 2026-08-27 (despliegue de `escudoverde-site`)
### 10.1 `/deploy` ahora exige POST, no GET
El §8.4 dice `GET /deploy?uuid=&force=true`. **Hoy responde 405 Method Not Allowed.**
Con `-Method POST` funciona y devuelve el `deployment_uuid` normalmente:
```powershell
Invoke-RestMethod "$env:COOLIFY_API_URL/deploy?uuid=<app_uuid>&force=true" -Headers $h -Method POST
```
`GET /deployments/{uuid}` sigue funcionando igual para el seguimiento.
### 10.2 El `GITHUB_TOKEN` de `.env.local.ps1` está caducado
`New-GitHubRepo.ps1` falla con `401 Bad credentials`. El token que **sí** sirve es el que
guarda el Administrador de credenciales de Windows para `github.com` (cuenta
`urieljarethbusiness-cpu`, scopes `gist, repo, workflow`). Se recupera sin exponerlo:
```bash
TOK=$(printf "protocol=https\nhost=github.com\n\n" | git credential fill | sed -n 's/^password=//p')
```
Conviene rotar el de `.env.local.ps1` o hacer que los scripts caigan a `git credential fill`.
### 10.3 Imágenes nginx sin root: el orden dentro del `RUN` importa
El builder de Coolify corre el `docker build` **sin DAC override para root**. Consecuencia
concreta con la receta habitual de nginx no-root:
- Si `nginx -t` va **después** del `chown` de `/tmp` al usuario `nginx`, el build falla con
`open() "/tmp/nginx.pid" failed (13: Permission denied)` — aunque en local funcione.
- `nginx -t` **crea** el fichero pid. Si se queda en la imagen, pertenece a root y el
contenedor no arranca al hacer `USER nginx`.
Orden que funciona en ambos lados:
```dockerfile
RUN set -eux; \
mkdir -p /tmp/nginx/client_body /tmp/nginx/proxy; \
nginx -t -c /etc/nginx/nginx.conf; \
rm -f /tmp/nginx/nginx.pid; \
chown -R nginx:nginx /tmp/nginx /usr/share/nginx/html /var/cache/nginx; \
chmod 0777 /tmp/nginx
USER nginx
```
### 10.4 `immutable` + nombre de fichero fijo = despliegue invisible
Con `Cache-Control: public, max-age=31536000, immutable` sobre `/assets/`, Cloudflare
sirvió el CSS anterior durante todo el despliegue siguiente (`cf-cache-status: HIT`,
`Age: 1008`). El contenedor tenía la versión nueva; el usuario veía la vieja.
**Regla para cualquier app estática en esta instancia:** o los assets llevan hash en el
nombre, o el HTML los referencia con `?v=<hash del contenido>`. Sin eso, `immutable` es
una trampa: el despliegue "funciona" y no cambia nada visible.
---
## 11. Re-verificación completa 2026-08-29 (v4.3.14): el 404 de `/applications` era Cloudflare
Auditoría integral con SSH + APIs validadas. La instancia corre ahora
**`4.3.14`** (`GET /version`), actualizada desde 4.3.10. Hallazgos, en orden de
importancia:
### 11.1 La API está COMPLETA — el 404 de `/applications/*` lo impone Cloudflare, no Coolify
Mismo token de siempre, misma ruta, dos caminos:
| Llamada | Resultado |
|---|---|
| `https://coolify.urieljareth.org/api/v1/applications` (público, vía Cloudflare) | **404** |
| `http://192.168.0.117:8000/api/v1/applications` (origen, LXC 102) | **200** (JSON completo) |
| `/github-apps` | 404 vía CF · **200 vía origen** |
| `/dockerfiles`, `/sources`, `/notifications`, `/private-keys` | 404 en ambos → **no existen como rutas**; los nombres correctos están en el spec (ver 11.2) |
| `/version`, `/resources`, `/servers`, `/projects`, `/teams`, `/services`, `/databases`, `/deployments`, `/security/keys` | 200 por ambas vías |
Todo el diagnóstico de §1 ("API parcial", v4.1.2) era este mismo bloqueo de
Cloudflare, no un recorte de la API. Las secciones §3 (Playwright) y §8 (DB
INSERT) siguen siendo fallbacks válidos, pero **la vía API debería funcionar
llamando al origen** (`$env:COOLIFY_API_URL_ORIGIN`). No se ha ejercitado
end-to-end un `POST /applications/*` contra el origen todavía — verificarlo en el
próximo deploy antes de jubilar el flujo UI. El fix de raíz es corregir la regla
del edge en el dashboard de Cloudflare.
### 11.2 La superficie real (del `openapi.yaml` del propio contenedor)
Dentro del contenedor: `/var/www/html/openapi.yaml`. Rutas declaradas en
4.3.14 (extraídas el 2026-08-29):
- `/applications` + `/applications/{public, private-github-app, private-deploy-key, dockerfile, dockerimage}`
- `/databases` + por motor: `{postgresql, clickhouse, dragonfly, redis, keydb, mariadb, mysql, mongodb}`
- `/deployments`, `/deploy`, `/destinations`, `/servers` (+ `/import`, `/digitalocean`, `/hetzner`, `/vultr`)
- `/github-apps`, `/gitlab-apps`, `/security/keys`, `/s3-storages`, `/tags`
- `/notifications/{email, discord, slack, telegram, pushover, webhook}`
- `/cloud-init-scripts`, `/cloud-tokens`
- `/projects`, `/projects/{uuid}/environments`, `/resources`, `/services`
- `/team`, `/teams`, `/team/envs`, `/team/members`
- `/version`, `/health`, `/enable`, `/disable`, `/mcp/enable`, `/mcp/disable`
`GET /docs` (Scalar UI) redirige a `/login` — requiere sesión web; para el
contrato exacto, leer el YAML del contenedor:
`pct exec 102 -- docker exec coolify cat /var/www/html/openapi.yaml`.
### 11.3 POST-only confirmado como comportamiento de Coolify (no de Cloudflare)
Contra el origen, `GET /deploy?uuid=fake` → **405**
`{"message":"This endpoint has changed to a POST request."}`. Es un cambio real
de Coolify v4.2 (changelog: los endpoints de estado pasaron a POST-only), igual
que §10.1. Otros cambios relevantes de 4.2→4.3: endpoints de logs para
db/servicio/contenedor, settings de application incluidas en las respuestas,
soporte MCP (read-only) y build pack Railpack (beta, 4.3.1). Breaking: se
eliminó el endpoint deprecado de aplicación Docker Compose — usar
`POST /services`. Fuentes: coolify.io/changelog y
github.com/coollabsio/coolify/releases.
### 11.4 Estado de credenciales verificado el 2026-08-29
| Credencial | Estado |
|---|---|
| SSH `root@192.168.0.200` y `root@192.168.0.117` con `keys/proxmox_ed25519` (≡ `~/.ssh/coolify_key`, sin passphrase) | ✅ |
| API Proxmox: token `root@pam!openclaw` (único en `token.cfg`) | ✅ 200 `/version` + `/cluster/resources` — ya en `.env.local.ps1` |
| API Coolify: `COOLIFY_TOKEN` (team root) | ✅ 200 — v4.3.14 |
| Gitea: token de `urieljareth` | ✅ 200 `/api/v1/user` |
| GitHub: PAT `ghp_NFy4…` del `.env` | ❌ caducado (401) — reemplazado en `.env.local.ps1` por el token vivo del Credential Manager de Windows (`gho_…`, cuenta `urieljarethbusiness-cpu`, scopes `gist, repo, workflow`) |
| Cloudflare API | ❌ sin token — gestionar túnel desde el dashboard |
| UI Coolify: `urieljareth@gmail.com` / password habitual | ⚠️ probable (heredado de la instancia vieja `192.168.1.175`, ver `C:\Users\Uriel Jareth\coolify-agent\coolify-agent-skill.md`), sin verificar |
Inventario completo de llaves SSH de la máquina (incluidas las corruptas de
SiteGround y la de la instancia antigua): `ACCESS.md` en la raíz del repo.
@@ -7,7 +7,7 @@
$env:PROXMOX_HOST = "192.168.0.200" $env:PROXMOX_HOST = "192.168.0.200"
$env:PROXMOX_NODE = "thinkcentre" $env:PROXMOX_NODE = "thinkcentre"
$env:PROXMOX_USER = "root" $env:PROXMOX_USER = "root"
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" $env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519" # relativo a la raíz del repo (≡ ~\.ssh\coolify_key)
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json" $env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw" $env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET" $env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
@@ -16,10 +16,10 @@ $env:PROXMOX_COOLIFY_LXC = "102"
# Coolify # Coolify
$env:COOLIFY_API_URL = "https://coolify.urieljareth.org/api/v1" $env:COOLIFY_API_URL = "https://coolify.urieljareth.org/api/v1"
$env:COOLIFY_TOKEN = "REPLACE_WITH_COOLIFY_TOKEN" $env:COOLIFY_TOKEN = "REPLACE_WITH_COOLIFY_TOKEN"
# Coolify UI login (email/password) — REQUIRED for Playwright-driven UI operations. # Coolify UI login (email/password) — used by the Playwright UI fallback flow.
# This instance (v4.1.2) does NOT expose the /applications/* REST API (all 404), # Since v4.3.14 the REST API is complete when called against the origin
# so configuring a git-based / Docker-Compose application (build pack, env vars, # (http://192.168.0.117:8000/api/v1); the /applications/* 404 happens only via
# domains) can ONLY be done through the web UI. See deploy_skill/references/coolify-4.1.2-notes.md. # the public Cloudflare hostname. See deploy_skill/references/coolify-4.1.2-notes.md §11.
$env:COOLIFY_EMAIL = "REPLACE_WITH_COOLIFY_UI_EMAIL" $env:COOLIFY_EMAIL = "REPLACE_WITH_COOLIFY_UI_EMAIL"
$env:COOLIFY_PASSWORD = "REPLACE_WITH_COOLIFY_UI_PASSWORD" $env:COOLIFY_PASSWORD = "REPLACE_WITH_COOLIFY_UI_PASSWORD"
+6 -2
View File
@@ -29,7 +29,9 @@ $args = @("-sS", "-i", "-X", $Method.ToUpper(), "-H", "Authorization: Bearer $en
if ($PSBoundParameters.ContainsKey("BodyJson")) { if ($PSBoundParameters.ContainsKey("BodyJson")) {
try { $null = $BodyJson | ConvertFrom-Json } catch { throw "BodyJson is not valid JSON: $($_.Exception.Message)" } try { $null = $BodyJson | ConvertFrom-Json } catch { throw "BodyJson is not valid JSON: $($_.Exception.Message)" }
$tmpBody = [System.IO.Path]::GetTempFileName() $tmpBody = [System.IO.Path]::GetTempFileName()
Set-Content -LiteralPath $tmpBody -Value $BodyJson -NoNewline -Encoding utf8 # Set-Content -Encoding utf8 mete BOM en PS 5.1 y GitHub responde
# 400 "Problems parsing JSON". WriteAllText con UTF8Encoding($false) no lo hace.
[System.IO.File]::WriteAllText($tmpBody, $BodyJson, (New-Object System.Text.UTF8Encoding($false)))
$args += @("--data", "@$tmpBody", "-H", "Content-Type: application/json") $args += @("--data", "@$tmpBody", "-H", "Content-Type: application/json")
} }
@@ -44,7 +46,9 @@ try {
$bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" } $bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" }
$statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim() $statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim()
if ($statusLine -notmatch " 20[04-9] ") { # 2xx completo: crear un repo responde 201 y la clase anterior ([04-9]) lo
# trataba como error pese al éxito.
if ($statusLine -notmatch " 20[0-9] ") {
$snippet = $bodyBlock.Trim() $snippet = $bodyBlock.Trim()
if ($snippet.Length -gt 400) { $snippet = $snippet.Substring(0, 400) + "..." } if ($snippet.Length -gt 400) { $snippet = $snippet.Substring(0, 400) + "..." }
throw "GitHub API HTTP error: $statusLine`nBody: $snippet" throw "GitHub API HTTP error: $statusLine`nBody: $snippet"
@@ -0,0 +1,110 @@
<#
.SYNOPSIS
Create a Coolify application directly in the database (bypasses the 404 API and the web UI).
.DESCRIPTION
For instances where POST /applications/* returns 404 and no UI credentials are available.
Requires SSH access to the Proxmox host and the Coolify DB container.
Creates:
1. applications row (Dockerfile build pack, GitHub source)
2. application_settings row (prevents "disable_build_cache on null" crash)
Then triggers a deploy via the API.
See deploy_skill/references/coolify-4.1.2-notes.md §8 for the full backstory.
.PARAMETER AppName
Display name (also used in the generated UUID suffix).
.PARAMETER GitRepo
GitHub repo in owner/repo format (e.g. urieljarethbusiness-cpu/agendamax).
.PARAMETER Fqdn
Full domain (e.g. https://agendamax.urieljareth.org).
.PARAMETER Port
Internal port the app listens on. Default: 3000.
.PARAMETER EnvironmentId
Numeric ID of the Coolify environment. Find with:
SELECT id FROM environments WHERE uuid = '<env_uuid>';
.PARAMETER GithubRepoId
Numeric GitHub repo ID. Get from: GET https://api.github.com/repos/<owner>/<repo> → .id
.EXAMPLE
. .\.env.local.ps1
.\deploy_skill\scripts\New-CoolifyAppViaDB.ps1 `
-AppName agendamax `
-GitRepo urieljarethbusiness-cpu/agendamax `
-Fqdn https://agendamax.urieljareth.org `
-Port 3000 -EnvironmentId 9 -GithubRepoId 1314049704
#>
param(
[Parameter(Mandatory)] [string]$AppName,
[Parameter(Mandatory)] [string]$GitRepo,
[Parameter(Mandatory)] [string]$Fqdn,
[int]$Port = 3000,
[Parameter(Mandatory)] [int]$EnvironmentId,
[Parameter(Mandatory)] [int64]$GithubRepoId
)
$ErrorActionPreference = "Stop"
. "$PSScriptRoot\..\..\scripts\ProxmoxAgent.ps1"
# Generate a 24-char lowercase UUID (Coolify style)
$uuid = -join ((1..24) | ForEach-Object { '{0:x}' -f (Get-Random -Max 16) })
Write-Host "Generated UUID: $uuid" -ForegroundColor Cyan
# Build the SQL (single-line to avoid heredoc issues through SSH)
$sql = @"
INSERT INTO applications (uuid, name, git_repository, git_branch, git_commit_sha, build_pack, dockerfile_location, ports_exposes, health_check_path, health_check_port, health_check_host, health_check_method, health_check_return_code, health_check_scheme, health_check_interval, health_check_timeout, health_check_retries, health_check_start_period, health_check_enabled, limits_memory, limits_memory_swap, limits_memory_swappiness, limits_memory_reservation, limits_cpus, limits_cpu_shares, status, preview_url_template, fqdn, repository_project_id, source_type, source_id, destination_type, destination_id, environment_id, base_directory, static_image, created_at, updated_at) VALUES ('${uuid}', '${AppName}:main-${uuid}', '${GitRepo}', 'main', 'HEAD', 'dockerfile', '/Dockerfile', '${Port}', '/', '${Port}', 'localhost', 'GET', 200, 'http', 5, 5, 10, 5, false, '0', '0', 60, '0', '0', 1024, 'exited', '{{pr_id}}.{{domain}}', '${Fqdn}', ${GithubRepoId}, 'App\Models\GithubApp', 3, 'App\Models\StandaloneDocker', 0, ${EnvironmentId}, '/', 'nginx:alpine', NOW(), NOW()) RETURNING id;
"@
# Write the SQL to a temp .sh script, scp it, execute it
$tmpSh = [IO.Path]::GetTempFileName() + ".sh"
@"
pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -t -A -c "$($sql -replace '"', '\"' -replace "'", "'\''")"
"@ | Set-Content -Path $tmpSh -Encoding ASCII
Write-Host "`n>>> Step 1: INSERT application row" -ForegroundColor Magenta
$remoteSh = "/tmp/coolify_create_$(Get-Random).sh"
scp -o BatchMode=yes -o StrictHostKeyChecking=no -i $env:PROXMOX_SSH_KEY $tmpSh "root@$($env:PROXMOX_HOST):$remoteSh" 2>&1 | Out-Null
$appId = Invoke-ProxmoxSshCommand -Command "sh $remoteSh; rm $remoteSh"
$appId = $appId.Trim()
Remove-Item $tmpSh -ErrorAction SilentlyContinue
Write-Host "Application ID: $appId" -ForegroundColor Green
if (-not $appId -or $appId -notmatch '^\d+$') {
throw "INSERT failed or didn't return an ID. Output: $appId"
}
# Step 2: application_settings
Write-Host "`n>>> Step 2: INSERT application_settings row" -ForegroundColor Magenta
$sqlSettings = @"
INSERT INTO application_settings (application_id, is_static, is_git_submodules_enabled, is_git_lfs_enabled, is_auto_deploy_enabled, is_force_https_enabled, is_debug_enabled, is_preview_deployments_enabled, is_log_drain_enabled, is_gpu_enabled, is_swarm_only_worker_nodes, is_raw_compose_deployment_enabled, is_build_server_enabled, is_consistent_container_name_enabled, is_gzip_enabled, is_stripprefix_enabled, is_container_label_escape_enabled, is_container_label_readonly_enabled, disable_build_cache, is_spa, is_git_shallow_clone_enabled, is_pr_deployments_public_enabled, use_build_secrets, inject_build_args_to_dockerfile, docker_images_to_keep, created_at, updated_at) VALUES (${appId}, false, true, true, true, true, false, false, false, false, true, false, false, false, true, true, true, true, false, false, true, false, false, true, 2, NOW(), NOW());
"@
$tmpSh2 = [IO.Path]::GetTempFileName() + ".sh"
@"
pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -c "$($sqlSettings -replace '"', '\"' -replace "'", "'\''")"
"@ | Set-Content -Path $tmpSh2 -Encoding ASCII
$remoteSh2 = "/tmp/coolify_settings_$(Get-Random).sh"
scp -o BatchMode=yes -o StrictHostKeyChecking=no -i $env:PROXMOX_SSH_KEY $tmpSh2 "root@$($env:PROXMOX_HOST):$remoteSh2" 2>&1 | Out-Null
Invoke-ProxmoxSshCommand -Command "sh $remoteSh2; rm $remoteSh2" | Write-Host
Remove-Item $tmpSh2 -ErrorAction SilentlyContinue
Write-Host "Settings created for app_id=$appId" -ForegroundColor Green
# Step 3: Deploy via API
Write-Host "`n>>> Step 3: Trigger deploy" -ForegroundColor Magenta
$headers = @{ Authorization = "Bearer $($env:COOLIFY_TOKEN)" }
$deploy = Invoke-RestMethod "$($env:COOLIFY_API_URL)/deploy?uuid=${uuid}&force=true" -Headers $headers
$deployUuid = $deploy.deployments[0].deployment_uuid
Write-Host "Deployment queued: $deployUuid" -ForegroundColor Green
Write-Host "`n=== DONE ===" -ForegroundColor Green
Write-Host "App UUID: $uuid"
Write-Host "App ID: $appId"
Write-Host "FQDN: $Fqdn"
Write-Host "Deploy: $deployUuid"
Write-Host "Monitor: GET $($env:COOLIFY_API_URL)/deployments/$deployUuid"
+11 -6
View File
@@ -213,13 +213,13 @@ Write-Host "Using server: $serverUuid" -ForegroundColor Cyan
if ($ServiceUuid) { if ($ServiceUuid) {
Write-Host "Updating existing service: $ServiceUuid" -ForegroundColor Cyan Write-Host "Updating existing service: $ServiceUuid" -ForegroundColor Cyan
# PATCH /services/{uuid} rechaza (422 "This field is not allowed")
# server_uuid/project_uuid/environment_name — solo se envian name,
# docker_compose_raw y urls. Verificado contra la instancia el 2026-09-02.
$patch = [ordered]@{ $patch = [ordered]@{
name = $AppName name = $AppName
docker_compose_raw = $composeRaw docker_compose_raw = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($composeRaw))
urls = $urlsList urls = $urlsList
project_uuid = $proj.uuid
environment_name = $env.name
server_uuid = $serverUuid
} }
if (-not (Confirm-Step "PATCH /services/$ServiceUuid (compose=$($composePath), urls=$($urlsList.Count), fqdn=$Fqdn)")) { if (-not (Confirm-Step "PATCH /services/$ServiceUuid (compose=$($composePath), urls=$($urlsList.Count), fqdn=$Fqdn)")) {
throw "Aborted by user." throw "Aborted by user."
@@ -227,13 +227,18 @@ if ($ServiceUuid) {
$result = & $coolifyApi -Method PATCH -Path "/services/$ServiceUuid" -BodyJson ($patch | ConvertTo-Json -Depth 6) -Raw -ErrorAction Stop $result = & $coolifyApi -Method PATCH -Path "/services/$ServiceUuid" -BodyJson ($patch | ConvertTo-Json -Depth 6) -Raw -ErrorAction Stop
$svcUuid = $ServiceUuid $svcUuid = $ServiceUuid
} else { } else {
# Verified against Coolify 4.x on 2026-08-24:
# - Do NOT send `type` together with `docker_compose_raw`. The API answers
# 422 "You cannot provide both service type and docker_compose_raw."
# `type` is only for one-click services from the library.
# - `docker_compose_raw` MUST be base64. Sent raw it answers
# 422 "The docker_compose_raw should be base64 encoded."
$createBody = [ordered]@{ $createBody = [ordered]@{
type = "one-click-service" # custom service; Coolify accepts any non-empty string here
name = $AppName name = $AppName
project_uuid = $proj.uuid project_uuid = $proj.uuid
environment_name = $env.name environment_name = $env.name
server_uuid = $serverUuid server_uuid = $serverUuid
docker_compose_raw = $composeRaw docker_compose_raw = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($composeRaw))
urls = $urlsList urls = $urlsList
instant_deploy = [bool]$InstantDeploy instant_deploy = [bool]$InstantDeploy
} }
+6 -6
View File
@@ -94,7 +94,7 @@ Each rule lists **what**, **why**, and **how to verify**.
- **Why:** TLS is issued by Traefik via **DNS challenge** (Cloudflare API token), so it - **Why:** TLS is issued by Traefik via **DNS challenge** (Cloudflare API token), so it
works regardless of Cloudflare's "Always Use HTTPS". Do **not** assume HTTP-01 challenge works regardless of Cloudflare's "Always Use HTTPS". Do **not** assume HTTP-01 challenge
— that path is intentionally not used here — that path is intentionally not used here
(see [`issue-coolify-static-app-deploy.md`](issue-coolify-static-app-deploy.md)). (see [`2026-04-11-coolify-static-app-deploy.md`](incidentes/2026-04-11-coolify-static-app-deploy.md)).
- **Verify:** - **Verify:**
```powershell ```powershell
curl.exe -k -sSI https://<name>.urieljareth.org/ curl.exe -k -sSI https://<name>.urieljareth.org/
@@ -285,13 +285,13 @@ curl.exe -k -sSI https://<name>.urieljareth.org/
| Symptom | Root cause | Fix | Source | | Symptom | Root cause | Fix | Source |
|---------|-----------|-----|--------| |---------|-----------|-----|--------|
| App's domain returns **502 Bad Gateway** | `ports_exposes` doesn't match the real listening port (often `3000` vs `80`) | Set `ports_exposes` to the actual port before deploy | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) | | App's domain returns **502 Bad Gateway** | `ports_exposes` doesn't match the real listening port (often `3000` vs `80`) | Set `ports_exposes` to the actual port before deploy | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
| 502 / no certificate on the app domain | Traefik couldn't get a cert via HTTP challenge | Environment uses **DNS challenge**; ensure FQDN is set and don't depend on HTTP-01 | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) | | 502 / no certificate on the app domain | Traefik couldn't get a cert via HTTP challenge | Environment uses **DNS challenge**; ensure FQDN is set and don't depend on HTTP-01 | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
| App reachable only at an ugly **UUID subdomain** | FQDN not set, Coolify auto-generated it | Set `fqdn` to `https://<name>.urieljareth.org` before first deploy | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) | | App reachable only at an ugly **UUID subdomain** | FQDN not set, Coolify auto-generated it | Set `fqdn` to `https://<name>.urieljareth.org` before first deploy | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
| App **won't start**, port conflict | Compose publishes `80`/`443` (owned by `coolify-proxy`) | Remove host port publishing; let Traefik route | [runbooks/baserow.md](runbooks/baserow.md) | | App **won't start**, port conflict | Compose publishes `80`/`443` (owned by `coolify-proxy`) | Remove host port publishing; let Traefik route | [runbooks/baserow.md](runbooks/baserow.md) |
| App **can't reach its DB** | Used `localhost`, or DB is on a different network | Use the DB **service name**; for shared services join the `coolify` network | [runbooks/nextcloud.md](runbooks/nextcloud.md), [runbooks/baserow.md](runbooks/baserow.md) | | App **can't reach its DB** | Used `localhost`, or DB is on a different network | Use the DB **service name**; for shared services join the `coolify` network | [runbooks/nextcloud.md](runbooks/nextcloud.md), [runbooks/baserow.md](runbooks/baserow.md) |
| Coolify **UI blank** when opening the app page | Cloudflare tunnel route order — `/app/*` captured `/application/...` | Operator fix: `/project/*` route must precede `/app/*` in the dashboard | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) | | Coolify **UI blank** when opening the app page | Cloudflare tunnel route order — `/app/*` captured `/application/...` | Operator fix: `/project/*` route must precede `/app/*` in the dashboard | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
| **WebSocket / terminal** drops, `tls: first record does not look like a TLS handshake` | Tunnel routes for ports `6001`/`6002` set to `https://` | Operator fix: those routes must be `http://` | [ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md](ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md) | | **WebSocket / terminal** drops, `tls: first record does not look like a TLS handshake` | Tunnel routes for ports `6001`/`6002` set to `https://` | Operator fix: those routes must be `http://` | [2026-04-11-cloudflare-tunnel-websocket-tls.md](incidentes/2026-04-11-cloudflare-tunnel-websocket-tls.md) |
> The last two are *operator/infrastructure* fixes (Cloudflare dashboard), not things the > The last two are *operator/infrastructure* fixes (Cloudflare dashboard), not things the
> app developer changes — listed here so an agent recognizes the symptom and points the > app developer changes — listed here so an agent recognizes the symptom and points the
+408
View File
@@ -0,0 +1,408 @@
# Tool index — catálogo canónico de herramientas
**Este es el índice único de todo lo ejecutable del repo.** Si buscas "qué script
uso para X", empieza aquí y no en los `TOOLS.md` de cada skill (esos son guías de
uso; este es el catálogo).
Verificado contra el host real el **2026-08-07**. Las firmas de parámetros se
extrajeron del AST de PowerShell, no a mano.
---
## 0. Cómo leer este índice
- **R/W** — `RO` = solo lectura, se puede ejecutar sin preguntar. `W` = muta
estado, **exige confirmación explícita del usuario antes de ejecutar**.
`RO/W` = depende de los parámetros (columna "notas" lo aclara).
- **Env** — variables que deben estar cargadas (`. .\.env.local.ps1`). Si faltan,
el script *lanza excepción*, no falla silenciosamente.
- Todo se ejecuta desde la raíz del repo, en PowerShell.
---
## 1. Los 6 gotchas que producen resultados silenciosamente incorrectos
Léelos antes de invocar nada. No son teóricos: los cuatro primeros se
verificaron el 2026-08-07 y el quinto el 2026-08-23. Cada uno rompe una tarea de
forma que *parece* haber funcionado.
### 1.1 `-Raw` significa lo OPUESTO en dos familias de wrappers
Hay dos convenciones incompatibles. Elegir mal no da error: da un resultado vacío.
| Wrapper | Sin `-Raw` (default) | Con `-Raw` |
|---|---|---|
| `coolify_skill\scripts\Invoke-CoolifyApi.ps1` | **string JSON** | **objetos PowerShell** |
| `scripts\Invoke-CloudflareApi.ps1` | **string JSON** | **objetos PowerShell** |
| `deploy_skill\scripts\Invoke-GitHubApi.ps1` | **objetos PowerShell** | `{Status, Headers, Body}` |
| `gitea_skill\scripts\Invoke-GiteaApi.ps1` | **objetos PowerShell** | `{Status, Headers, Body}` |
Consecuencia práctica: en Coolify y Cloudflare, **si vas a filtrar o proyectar el
resultado necesitas `-Raw`**. Sin él recibes un `System.String` y
```powershell
# MAL: devuelve una fila vacía, sin error. El resultado es un string.
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" |
Select-Object name, uuid
# BIEN
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
Select-Object name, uuid
```
Usa el default (string JSON) solo cuando vas a mostrar la respuesta tal cual.
### 1.2 `Invoke-ProxmoxSsh.ps1` corrompe las comillas anidadas
`Invoke-ProxmoxSshCommand` pasa `$Command` como un único argumento a `ssh`, y
PowerShell 5.1 destroza las comillas embebidas al invocar un ejecutable nativo.
Cualquier comando con quoting anidado —típicamente `docker exec ... bash -lc '...
psql -c "SELECT ..."'`— llega mutilado al host:
```
bash: line 1: -c: command not found
psql: option requires an argument -- 'F'
```
**Solución verificada: codifica el comando remoto en base64.** Es el patrón a
usar para cualquier cosa con más de un nivel de comillas:
```powershell
$remote = @'
pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -c "SELECT name FROM public.installation_configs"'
'@
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($remote))
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "echo $b64 | base64 -d | bash 2>&1"
```
El here-string `@'...'@` (comillas simples) es obligatorio: evita que PowerShell
expanda `$POSTGRES_PASSWORD` del lado de Windows.
Para comandos de un solo nivel de comillas (`pct list`, `docker ps --format
'{{.Names}}'`) el wrapper directo funciona bien.
### 1.3 Los nombres de contenedor NO se pueden adivinar
Coolify nombra cada contenedor `<servicio>-<uuid>` (y a veces le añade un sufijo
numérico de build). No existe un contenedor llamado `chatwoot` ni `nextcloud`:
```
chatwoot-c11xzy2tx2cdapm32f5b89vy
nextcloud-db-hdcdpkm0jko3qqvn5683ercc
web-instademo0portal0insta0demo1-060825532589
```
**Siempre resuelve el nombre real antes de usarlo.** Dos caminos:
```powershell
# a) desde Docker, por patrón
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i chatwoot"
# b) desde Coolify, para obtener el uuid del recurso (y de ahí el sufijo)
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
Where-Object { $_.name -match 'chatwoot' } | Select-Object name, uuid, fqdn
```
Corolario: **un redeploy puede recrear el contenedor con otro sufijo**, y cualquier
script o cron que tenga el nombre hardcodeado empieza a fallar. Es exactamente lo
que le pasó al guard de Chatwoot (ver [runbooks/chatwoot-update.md](runbooks/chatwoot-update.md)).
### 1.4 `/applications/*` da 404 — por Cloudflare, no por Coolify (corregido 2026-08-29)
**Re-verificado a fondo el 2026-08-29 sobre v4.3.14 y el diagnóstico anterior
cambió:** el 404 NO lo produce Coolify. Es un bloqueo del edge de Cloudflare en
el hostname público. Mismo token, misma ruta:
| Llamada | Resultado |
|---|---|
| `https://coolify.urieljareth.org/api/v1/applications` (vía Cloudflare) | **404** |
| `http://192.168.0.117:8000/api/v1/applications` (origen, LXC 102) | **200** |
| `/github-apps` | igual: 404 vía CF, 200 vía origen |
| `/version`, `/resources`, `/services`, `/databases`, `/projects`, `/servers`, `/teams`, `/deployments`, `/security/keys` | OK por ambas vías |
La API del origen está **completa**: el `openapi.yaml` del contenedor
(`/var/www/html/openapi.yaml`) declara todo el namespace de `/applications/*`,
notificaciones, proveedores cloud y MCP, y responde. **Para esos endpoints,
apunta la llamada al origen** (`$env:COOLIFY_API_URL_ORIGIN`) o corrige la regla
de Cloudflare en el dashboard. Además, desde v4.2 los endpoints de estado
exigen **POST** (`GET /deploy` → 405; ver notas §10.1).
Esto re-habilita (previa verificación en el próximo deploy) herramientas que
estaban marcadas rotas — ver §4.
**Nota (2026-08-24):** `POST /services` **sí funciona** para stacks compose
propios, con tres condiciones que la API no perdona: **no** enviar `type` junto a
`docker_compose_raw`, mandar el compose en **base64**, y enviar el body como
**bytes UTF-8**. Las tres estaban mal en el toolkit y ya están corregidas; el
detalle está en
[docs/casos/firecrawl-stack-minimo.md §4](casos/firecrawl-stack-minimo.md).
`Test-PreDeployChecklist.ps1` sigue sin leer `docker-compose.coolify.yml`
(solo mira `docker-compose.yml|yaml` y `compose.yml|yaml`), así que **puede dar
todo PASS sobre un archivo que no abrió**.
### 1.5 "Verde en Coolify" NO significa "enrutado en Traefik"
Verificado el 2026-08-23. Es la causa de que un servicio recién creado desde la
librería de Coolify devuelva **`503 no available server`** estando en verde.
Son dos señales distintas y la UI solo muestra una:
| Señal | Quién la usa | Qué significa |
|---|---|---|
| `State.Status = running` | **La UI de Coolify** (el punto verde) | El contenedor existe y no ha muerto |
| `State.Health.Status = healthy` | **Traefik** | El contenedor entra al balanceador |
Traefik solo enruta contenedores que Docker reporta `healthy`. Un contenedor
`running` + `unhealthy` **no tiene ruta**, la petición cae al catch-all de
Coolify (`default_redirect_503.yaml`: `priority: -1000`, servicio `noop` con
`servers: { }`) y de ahí sale el string `no available server`.
En este host el primer boot de un servicio tarda **minutos** (rootfs ext4 sobre
loopback sobre HDD, ~39 ms por escritura), pero **12 de los 14 servicios no
tienen `start_period`** en su healthcheck. Se les declara `unhealthy` mucho antes
de que la app llegue a escuchar.
**Distingue el código HTTP antes de tocar nada:**
- **`502 Bad Gateway`** → Traefik *tiene* la ruta, el backend rechaza. Problema
de la app o del puerto.
- **`503 no available server`** → Traefik **no tiene** la ruta. Casi siempre es
un contenedor que todavía está arrancando. **Espera, no redeployes**: un
redeploy reinicia el entrypoint desde cero y reinicia el arranque lento.
```powershell
# El diagnóstico correcto, en un comando (solo lectura):
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
```
Detalle completo, evidencia e hipótesis descartadas:
[docs/casos/coolify-servicio-nuevo-503-no-available-server.md](casos/coolify-servicio-nuevo-503-no-available-server.md).
### 1.6 Si la imagen no expone el puerto donde sirve, el FQDN necesita `:puerto`
Verificado el 2026-08-24 diagnosticando `grimmory`, que devolvía **502**.
Coolify deriva el label `traefik.http.services.*.loadbalancer.server.port` del
**puerto que lleve el FQDN guardado** en `service_applications.fqdn`. Si el FQDN
no lleva puerto, Coolify **no emite el label**, y Traefik cae al único puerto que
declare la imagen (`Config.ExposedPorts`).
Eso funciona por accidente cuando app e imagen coinciden, y falla en silencio
cuando no:
| Servicio | FQDN guardado | Sirve en | Imagen expone | Resultado |
|---|---|---|---|---|
| `nextcloud` | sin puerto | 80 | 80 | OK por coincidencia |
| `n8n` | `…:5678` | 5678 | 5678 | OK explícito |
| `qdrant` | `…:6333` | 6333 | 6333 | OK explícito |
| **`grimmory`** | **sin puerto** | **80** | **6060** | **502** |
`grimmory` corría `healthy` (su healthcheck prueba `http://127.0.0.1/health`, o
sea el 80, y pasaba), Traefik lo enrutaba, y llegaba al 6060 donde no hay nada:
**connection refused → 502**.
Ojo con la confusión: tener `SERVICE_URL_GRIMMORY_80` en el compose **no basta**.
Lo que manda es el puerto en el FQDN almacenado.
**Cómo distinguirlo de otros fallos:**
- **`502`** → hay ruta, el backend rechaza. Compara el puerto donde escucha la app
con el que busca Traefik. Casi siempre es esto.
- **`503 no available server`** → no hay ruta (§1.5).
```powershell
# Puertos donde escucha de verdad (en hex; 0050 = 80, 1F90 = 8080):
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'cat /proc/net/tcp'"
# Puerto que busca Traefik (si no sale nada, cae al ExposedPorts de la imagen):
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker inspect <cont> --format '{{json .Config.Labels}}'"
```
**Arreglo:** poner el puerto en el FQDN (`https://x.urieljareth.org:80`) y
**redeployar** el servicio — los labels solo se regeneran al recrear el
contenedor.
---
## 2. Infraestructura: host Proxmox, red, apps del host
Todo llega al host por un solo camino:
```
PowerShell → Invoke-ProxmoxSshCommand → ssh [email protected]
→ pct exec 102 -- docker ... (para cualquier cosa de Docker)
```
| Script | R/W | Env | Parámetros | Qué hace |
|---|---|---|---|---|
| [scripts/ProxmoxAgent.ps1](../scripts/ProxmoxAgent.ps1) | librería | — | — | **Dot-source obligatorio** (`. .\scripts\ProxmoxAgent.ps1`). Expone `Get-ProxmoxConfig`, `Assert-ProxmoxConfig`, `Invoke-ProxmoxSshCommand`, `Invoke-ProxmoxApi`. Todos los demás scripts lo consumen; no reimplementes la conexión. |
| [scripts/Test-ProxmoxConnection.ps1](../scripts/Test-ProxmoxConnection.ps1) | RO | opcional `PROXMOX_API_TOKEN_*` | — | Smoke test: config + SSH + muestra de Docker + auth de API. Sale 1 si algo falla. **Ejecútalo antes de trabajo operativo.** |
| [scripts/Get-ProxmoxInventory.ps1](../scripts/Get-ProxmoxInventory.ps1) | RO | — | — | Snapshot completo: host, LXC, QEMU, Docker en LXC 102. |
| [scripts/Invoke-ProxmoxSsh.ps1](../scripts/Invoke-ProxmoxSsh.ps1) | **RO/W** | — | `-Command <string>` | Comando arbitrario por SSH. **El R/W lo determina el comando**: `pct list` es RO, `pct stop` es W. Ver gotcha §1.2 para comillas anidadas. |
| [scripts/Invoke-CloudflareApi.ps1](../scripts/Invoke-CloudflareApi.ps1) | **RO/W** | `CLOUDFLARE_API_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Cloudflare (túnel + DNS). `GET` es RO; el resto W. `-Raw` → objetos (§1.1). |
| [scripts/Install-CoolifyAutostart.ps1](../scripts/Install-CoolifyAutostart.ps1) | **W** (RO con `-VerifyOnly`) | — | `-VerifyOnly` `-SkipOnboot` `-RunNow` `-Uninstall` | Instala el auto-arranque del stack tras corte de luz (LXC 102 + túnel). Usa `-VerifyOnly` para auditar sin tocar nada. |
**API REST de Proxmox** (distinta de la de Coolify): requiere
`PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`. `Invoke-ProxmoxApi` lanza
excepción si faltan.
> ⚠️ **Estado actual (2026-08-07):** `.env.local.ps1` **no** define
> `PROXMOX_API_TOKEN_ID`/`_SECRET`, `CLOUDFLARE_API_TOKEN` ni
> `COOLIFY_EMAIL`/`COOLIFY_PASSWORD`. Las tres rutas que dependen de ellas
> (API de Proxmox, API de Cloudflare, flujo UI de Coolify) **fallan hoy**. Lo
> que sí está cargado: `PROXMOX_HOST/NODE/USER/SSH_KEY/COOLIFY_LXC`,
> `COOLIFY_API_URL`, `COOLIFY_TOKEN`, `GITHUB_TOKEN`, `GITHUB_OWNER`,
> `GITEA_URL/USER/TOKEN`.
### 2.1 Chatwoot — parche enterprise
Contexto completo en [runbooks/chatwoot-update.md](runbooks/chatwoot-update.md).
| Script | R/W | Parámetros | Qué hace |
|---|---|---|---|
| [scripts/Get-ChatwootLicenseStatus.ps1](../scripts/Get-ChatwootLicenseStatus.ps1) | RO | `-ServiceUuid` `-AppContainer` `-DbContainer` `-Deep` | Estado de la licencia. `-Deep` verifica además los feature flags por cuenta (`accounts.feature_flags`). **Usa esto para diagnosticar; nunca el parche.** |
| [scripts/Apply-ChatwootEnterprisePatch.ps1](../scripts/Apply-ChatwootEnterprisePatch.ps1) | **W** | `-DryRun` `-ReenableAccountFeatures` `-ServiceUuid` `-Container` `-AppContainer` `-LxcId` `-ProxmoxHost` `-SshKey` | Reaplica el parche. **El SQL de 3 filas no basta**: pasa siempre `-ReenableAccountFeatures`. Usa `-DryRun` primero. |
| [scripts/chatwoot-enterprise-guard.sh](../scripts/chatwoot-enterprise-guard.sh) | — | — | **Copia versionada** del guard que corre en el host. No se ejecuta desde Windows. Instalado en `/root/scripts/chatwoot-enterprise-guard.sh`, agendado por `/etc/cron.d/chatwoot-enterprise-guard` cada 5 min. Log: `/var/log/chatwoot-enterprise-guard.log` (solo escribe cuando actúa). |
### 2.2 Artefactos que viven en el host (no se invocan desde Windows)
| Archivo | Qué es |
|---|---|
| [scripts/host/coolify-autostart.sh](../scripts/host/coolify-autostart.sh) | Script de arranque; lo despliega `Install-CoolifyAutostart.ps1`. |
| [scripts/host/coolify-autostart.service](../scripts/host/coolify-autostart.service) | Unit de systemd correspondiente. |
| [scripts/fix-nextcloud-config.php](../scripts/fix-nextcloud-config.php) | Fragmento puntual del fix HTTPS de Nextcloud. Ver [runbooks/nextcloud.md](runbooks/nextcloud.md). |
| [scripts/Set-ProxmoxEnv.example.ps1](../scripts/Set-ProxmoxEnv.example.ps1) | Plantilla de entorno. El template completo es [.env.example](../.env.example). |
### 2.3 Scripts de deploy específicos de una app
`scripts/apps/` guarda scripts one-off con uuid y dominio **hardcodeados**. Son
**mutantes** y están atados a un recurso concreto: lee la cabecera antes de
ejecutar uno, y verifica que el uuid siga siendo el correcto (§1.3).
| Script | R/W | Env | Qué hace |
|---|---|---|---|
| [scripts/apps/Deploy-SoloLeveling.ps1](../scripts/apps/Deploy-SoloLeveling.ps1) | **W** | `COOLIFY_*`, `PROXMOX_*`, `GITHUB_TOKEN` (scope `repo`) | Re-deploy de "El Sistema (Solo Leveling)" sin depender del pull de GHCR (el registry es privado y el token no tiene `read:packages`). Construye la imagen **en el servidor** (clone del repo privado + `docker build`), la taguea con el nombre que espera el compose generado por Coolify, y levanta el servicio. Params: `-ServiceUuid` `-Domain` `-Repo` `-Image` `-NoBuild`. Requiere que el service ya esté registrado en Coolify. |
| [scripts/apps/Deploy-OhDaddy.ps1](../scripts/apps/Deploy-OhDaddy.ps1) | **W** | `COOLIFY_*`, `PROXMOX_*` | Redeploy de oh-daddy (stack app+db+Inngest self-hosted, servicio `rzittzudkunwx8gilonn7tqe`). Clone del repo público + `stacks/oh-daddy/Dockerfile` inyectado, build de `oh-daddy-app:local` en el server, `compose up -d`, schema idempotente y re-registro de funciones Inngest (`PUT /api/inngest`). Params: `-ServiceUuid` `-Fqdn` `-Repo` `-Image` `-NoBuild`. Detalle: [casos/oh-daddy-deploy.md](casos/oh-daddy-deploy.md). |
---
## 3. Coolify — operación
| Script | R/W | Env | Parámetros | Qué hace |
|---|---|---|---|---|
| [coolify_skill/scripts/Get-CoolifyDockerStatus.ps1](../coolify_skill/scripts/Get-CoolifyDockerStatus.ps1) | RO | — | `-Filter <regex>` `-All` | Estado de contenedores vía LXC 102. `-All` lista todo; `-Filter` acota por regex. Primera parada para "¿está corriendo X?". |
| [coolify_skill/scripts/Test-CoolifyServiceReady.ps1](../coolify_skill/scripts/Test-CoolifyServiceReady.ps1) | RO | — | `-Uuid <uuid>` `-Fqdn <url>` `-WaitSeconds <n>` | Contrasta `running` vs `healthy` por contenedor (la discrepancia que la UI esconde), avisa de `start_period` insuficiente, detecta setup en curso (`chown`/`apt`) y prueba el dominio distinguiendo 503 de 502. **Primera parada para un `503 no available server`** — ver §1.5. |
| [coolify_skill/scripts/Set-CoolifyHealthcheckGrace.ps1](../coolify_skill/scripts/Set-CoolifyHealthcheckGrace.ps1) | **RO/W** | — | `-Uuid <uuid>` `-StartPeriodSeconds 300` `-MinIntervalSeconds 10` `-ShowResult` `-Apply` | Inserta `start_period` en los healthcheck que no lo tienen y sube intervalos demasiado cortos, editando `services.docker_compose_raw`. **Dry-run sin `-Apply`.** Con `-Apply` guarda copia de rollback en `backups/` y verifica leyendo de vuelta. **No redeploya** — el healthcheck solo aplica al recrear el contenedor. Ver §1.5. |
| [coolify_skill/scripts/Invoke-CoolifyApi.ps1](../coolify_skill/scripts/Invoke-CoolifyApi.ps1) | **RO/W** | `COOLIFY_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Coolify. `GET` RO; `POST/PUT/PATCH/DELETE` W → confirmación. Ver §1.1 (`-Raw`) y §1.4 (endpoints que dan 404). |
| `coolify_skill/scripts/coolify.sh` | RO/W | `COOLIFY_TOKEN` | — | Helper Bash legacy para sesiones Linux/WSL. En este repo se prefieren los wrappers PowerShell. |
**Referencia de la API:** no cargues el árbol completo. Busca y abre un solo
archivo:
```powershell
rg -n "deploy|database|environment" .\coolify_skill\references
Get-Content .\coolify_skill\references\ops\deploy-by-tag-or-uuid.md
```
---
## 4. Deploy de proyectos nuevos a Coolify
> ⚠️ **Lee esto antes de usar cualquier cosa de esta sección.** El pipeline
> "una sola línea" fallaba porque `/applications/*` daba 404 (§1.4). Desde el
> 2026-08-29 sabemos que ese 404 es de **Cloudflare, no de Coolify**: apuntando
> `COOLIFY_API_URL` al origen (`http://192.168.0.117:8000/api/v1`) el namespace
> completo responde. Scripts marcados ❌ abajo deben funcionar así — pendiente de
> verificar en el próximo deploy; el flujo UI y la vía DB siguen como fallback.
| Script | R/W | Env | Estado en esta instancia |
|---|---|---|---|
| [Publish-ProjectToCoolify.ps1](../deploy_skill/scripts/Publish-ProjectToCoolify.ps1) | W | `GITHUB_TOKEN`, `COOLIFY_TOKEN` | ⚠️ Su paso 5 invoca `New-CoolifyApplication.ps1` → probar contra el origen. |
| [New-CoolifyApplication.ps1](../deploy_skill/scripts/New-CoolifyApplication.ps1) | W | `COOLIFY_TOKEN` | ⚠️ Usa `POST /applications/public` y `PATCH /applications/{uuid}` → funcionan vía origen (404 solo vía Cloudflare). |
| [Invoke-CoolifyRollback.ps1](../deploy_skill/scripts/Invoke-CoolifyRollback.ps1) | W | `COOLIFY_TOKEN` | ⚠️ Usa `PATCH /applications/{uuid}` → funciona vía origen. |
| [New-CoolifyService.ps1](../deploy_skill/scripts/New-CoolifyService.ps1) | W | `COOLIFY_TOKEN` | ✅ **Funciona.** Usa `POST /services`. Es la vía válida para stacks multi-contenedor. |
| [New-CoolifyAppViaDB.ps1](../deploy_skill/scripts/New-CoolifyAppViaDB.ps1) | **W (INSERT directo en la DB)** | — | ⚠️ **Último recurso.** Escribe la fila en `applications` de la DB de Coolify saltándose API y UI. Sin validación ni rollback. Requiere `-EnvironmentId` y `-GithubRepoId` reales. |
| `coolify-ui/coolify-login.mjs` + `Configure-CoolifyComposeApp.mjs` | W (vía navegador) | `COOLIFY_EMAIL`, `COOLIFY_PASSWORD` | ✅ Fallback soportado para apps git build-from-source. Credenciales probables ya en `.env.local.ps1` (sin verificar). |
**Vías, en orden de preferencia:**
1. Stack multi-contenedor con `docker-compose.coolify.yml` → `New-CoolifyService.ps1` (API).
2. App git build-from-source → API contra el origen; si falla, flujo UI con Playwright (`coolify-ui/`).
3. Último recurso → `New-CoolifyAppViaDB.ps1`.
Detalle completo del flujo v4.1.2 y sus trampas (BOM UTF-8 rompe el parser YAML
de Coolify; "Reload Compose File" es obligatorio; fijar dominios por servicio o
sale 503; no encolar deploys concurrentes) en
[deploy_skill/references/coolify-4.1.2-notes.md](../deploy_skill/references/coolify-4.1.2-notes.md).
### 4.1 Scripts de deploy que funcionan sin depender de `/applications`
| Script | R/W | Env | Parámetros | Qué hace |
|---|---|---|---|---|
| [Initialize-CoolifyProject.ps1](../deploy_skill/scripts/Initialize-CoolifyProject.ps1) | W (solo local) | — | `-Path` `-Stack {node\|python\|compose\|static}` `-AppPort` `-Force` | Genera Dockerfile/compose/.dockerignore compatibles. Idempotente: no sobreescribe sin `-Force`. |
| [Test-PreDeployChecklist.ps1](../deploy_skill/scripts/Test-PreDeployChecklist.ps1) | RO | — | `-Path` `-ExpectedPort` `-Strict` | Valida el proyecto contra las reglas duras de [AGENTS-coolify-apps.md](AGENTS-coolify-apps.md). **Gate obligatorio antes de cualquier push.** |
| [New-GitHubRepo.ps1](../deploy_skill/scripts/New-GitHubRepo.ps1) | W (remoto) | `GITHUB_TOKEN` | `-Name` `-Description` `-Private` `-Owner` | Crea el repo en GitHub. Idempotente: avisa si ya existe. |
| [Invoke-GitHubApi.ps1](../deploy_skill/scripts/Invoke-GitHubApi.ps1) | RO/W | `GITHUB_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de GitHub. Default → objetos (§1.1). |
| [Test-PostDeploy.ps1](../deploy_skill/scripts/Test-PostDeploy.ps1) | RO | `COOLIFY_TOKEN` (opcional) | `-Fqdn` `-ContainerName` `-SiblingService` `-ApplicationUuid` | Verifica el estado en Coolify: contenedores, logs del proxy, DNS entre hermanos. `-ApplicationUuid` consulta `/applications/*` → vía pública da 404 (Cloudflare), vía origen funciona. |
| [Test-ServiceOnline.ps1](../deploy_skill/scripts/Test-ServiceOnline.ps1) | RO | — | `-Fqdn` `-Path` `-ExpectedStatusCode` `-ExpectTitle` `-Screenshot` `-SkipBrowser` `-TimeoutMs` `-Retries` | **La "definición de done".** Dos capas: HTTP 200 vía curl + render real en Chromium (Playwright). Detecta 502 de Traefik, TLS a medias y crashes de JS del cliente. Sin Node/Playwright la capa de navegador se omite con warning, no falla. |
| `deploy_skill/scripts/verify-online.mjs` | RO | — | (invocado por el script de arriba) | Capa de navegador de `Test-ServiceOnline.ps1`: navegación real con Playwright, captura `pageerror` y requests fallidos del main frame. No lo invoques directo. Requiere `npm install` en la raíz del repo. |
`git push` usa **Windows Credential Manager (wincred)**, no el PAT. Verificado
para la cuenta `urieljarethbusiness-cpu`.
**Plantillas:** [deploy_skill/references/templates/](../deploy_skill/references/templates/) —
`Dockerfile.node`, `Dockerfile.python`, `docker-compose.app-db.yml`,
`env.local.template.ps1`.
---
## 5. Gitea — hosting git self-hosted
Distinto de deploy_skill: esto administra la capa de hosting git, no Coolify.
El propio repo Manager vive aquí.
| Script | R/W | Env | Parámetros | Qué hace |
|---|---|---|---|---|
| [Test-GiteaConnection.ps1](../gitea_skill/scripts/Test-GiteaConnection.ps1) | RO | `GITEA_URL`, `GITEA_TOKEN` | — | Smoke test: `/version`, `/user`, `/settings/api`, `/repos/search`. Sale 1 si algo falla. |
| [Get-GiteaRepo.ps1](../gitea_skill/scripts/Get-GiteaRepo.ps1) | RO | `GITEA_URL`, `GITEA_TOKEN` | `-Owner` `-Name` `-Search` `-List` | Lee, lista o busca repos. `-Owner` default = usuario del token. |
| [Invoke-GiteaApi.ps1](../gitea_skill/scripts/Invoke-GiteaApi.ps1) | RO/W | `GITEA_URL`, `GITEA_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Gitea. Usa `curl.exe --data-binary` con UTF-8 sin BOM (el parser de Gitea se rompe con BOM). Default → objetos (§1.1). |
| [New-GiteaRepo.ps1](../gitea_skill/scripts/New-GiteaRepo.ps1) | W | `GITEA_URL`, `GITEA_TOKEN` | `-Name` `-Description` `-Private` `-Owner` `-NoAutoInit` | Crea repo (idempotente). `-NoAutoInit` evita el README del lado servidor para poder empujar historia local sin conflicto. |
| [Sync-GiteaRemote.ps1](../gitea_skill/scripts/Sync-GiteaRemote.ps1) | W | `GITEA_URL`, `GITEA_TOKEN` | `-AppPath` `-Name` `-Owner` `-RemoteName` `-Branch` `-CreateIfMissing` `-Private` `-Force` | Cablea el remoto y hace push headless. **El token nunca toca el disco**: va en un `http.extraHeader` de un solo uso, no en `.git/config`. `-RemoteName gitea` (default) convive con un `origin` de GitHub. |
---
## 6. Reglas de seguridad (no negociables)
1. **Read-only primero.** El default es diagnosticar: list, status, logs, inspect,
health checks.
2. **Confirmación explícita antes de cualquier cambio de estado.** Aplica a:
`pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose up-down;
deploys de Coolify y cualquier `POST`/`PUT`/`PATCH`/`DELETE`; escritura de
variables de entorno; y todo cambio de firewall, red, storage, volumen, clave
o token. Antes de una acción riesgosa: captura el estado actual y enuncia el
camino de rollback.
3. **Nunca escribas secretos en el repo.** Ni tokens, ni passwords, ni claves
privadas, ni cookies, ni secretos de token PVE — no en Markdown, no en
scripts, no en logs. Viven solo en `.env.local.ps1` (gitignored) o en el
almacén de secretos del SO. Al depurar bases de datos, verifica conectividad
sin imprimir credenciales.
4. **Prefiere los scripts del repo** antes que cadenas de comandos manuales
largas, y no construyas comandos destructivos amplios a partir de strings
generados.
5. `-Force` existe en varios scripts para saltarse los prompts. **Úsalo solo
después de que el usuario haya aprobado el plan concreto.**
---
## 7. Dónde está el resto del contexto
| Qué necesitas | Dónde |
|---|---|
| Arquitectura + router de intención | [CLAUDE.md](../CLAUDE.md) |
| Topología verificada (fuente de verdad del estado) | [proxmox-inventory.md](proxmox-inventory.md) |
| Procedimientos concretos | [runbooks/](runbooks/) |
| Casos resueltos paso a paso | [casos/](casos/) |
| Incidentes archivados | [incidentes/](incidentes/) |
| Contrato para *construir* una app deployable | [AGENTS-coolify-apps.md](AGENTS-coolify-apps.md) |
| Reglas operativas por dominio | `agent/SKILL.md`, `coolify_skill/SKILL.md`, `deploy_skill/SKILL.md`, `gitea_skill/SKILL.md` |
| Guías de uso por dominio (ejemplos ejecutables) | los `TOOLS.md` de cada carpeta `*_skill/` |
+49 -9
View File
@@ -1,7 +1,22 @@
# Caso: Parche enterprise en Chatwoot (Coolify + LXC 102) # Caso: Parche enterprise en Chatwoot (Coolify + LXC 102)
> Documentacion de caso verificada el 2026-06-16 desde esta maquina. > Documentacion de caso verificada el 2026-06-16 desde esta maquina.
> Dominio: `https://chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org` > Dominio: **`https://chat.urieljareth.org`** (corregido el 2026-07-24; el FQDN
> `chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org` que decia antes ya no aplica).
> **Leer antes de usar este caso (revision 2026-07-24):**
>
> 1. **El parche caduca solo en <= 24 h.** No hace falta actualizar para
> perderlo: `Internal::CheckNewVersionsJob` hace ping diario a
> `hub.2.chatwoot.com` y reescribe el plan con lo que responda el hub. Con el
> identifier actual la ventana es todos los dias a las **16:16 UTC**.
> 2. **Los 3 `UPDATE` de este caso no alcanzan** si el plan ya paso por
> `community`: `Internal::ReconcilePlanConfigService` apago los 9 feature
> flags premium en `accounts.feature_flags` de cada cuenta. Hay que correr
> `Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures`.
>
> Causa raiz completa, plan de actualizacion y fix durable:
> [docs/runbooks/chatwoot-update.md](../runbooks/chatwoot-update.md).
## 0. Resumen ejecutivo ## 0. Resumen ejecutivo
@@ -41,7 +56,7 @@
|---|---|---| |---|---|---|
| Host | Docker daemon local | Proxmox VE `192.168.0.200` | | Host | Docker daemon local | Proxmox VE `192.168.0.200` |
| Ejecucion Docker | `docker exec` directo | `pct exec 102 --` + `docker exec` | | Ejecucion Docker | `docker exec` directo | `pct exec 102 --` + `docker exec` |
| Usuario SSH | n/a | `root@192.168.0.200` con `~/.openclaw/workspace/proxmox_key_win` | | Usuario SSH | n/a | `root@192.168.0.200` con `keys/proxmox_ed25519` |
| Contenedor | filtro `name=pgvector` | nombre real: `postgres-c11xzy2tx2cdapm32f5b89vy` | | Contenedor | filtro `name=pgvector` | nombre real: `postgres-c11xzy2tx2cdapm32f5b89vy` |
| DB user / db | `-U postgres -d chatwoot` | `-U <POSTGRES_USER> -d <POSTGRES_DB>` autodetectados (imagen `pgvector/pgvector:pg12` **no crea rol `postgres`**) | | DB user / db | `-U postgres -d chatwoot` | `-U <POSTGRES_USER> -d <POSTGRES_DB>` autodetectados (imagen `pgvector/pgvector:pg12` **no crea rol `postgres`**) |
@@ -124,20 +139,29 @@ El script implementa el equivalente exacto y valida los `UPDATE 1`.
2. Sube 3 scripts `.sh` y un `.sql` al host Proxmox (no al LXC, para evitar 2. Sube 3 scripts `.sh` y un `.sql` al host Proxmox (no al LXC, para evitar
un `pct push` extra y problemas de ruta). un `pct push` extra y problemas de ruta).
3. Autodetecta: 3. Autodetecta:
- contenedor Postgres de Chatwoot por el patron - contenedor Postgres de Chatwoot: filtra `docker ps` por el uuid del
`c11xzy2tx2cdapm32f5b89vy.*(pgvector|postgres|db)`. servicio y luego por `(pgvector|postgres|db)`. Son **dos greps
encadenados** a proposito — Coolify nombra los contenedores
`<servicio>-<uuid>` (`postgres-c11xzy...`), asi que el patron unico
`<uuid>.*postgres` que tenia antes no casaba nunca y la autodeteccion
fallaba siempre (corregido el 2026-07-24).
- `POSTGRES_USER` / `POSTGRES_DB` / `POSTGRES_PASSWORD` desde - `POSTGRES_USER` / `POSTGRES_DB` / `POSTGRES_PASSWORD` desde
`docker inspect`. `docker inspect`.
4. Ejecuta el comando equivalente dentro del LXC, captura stdout y exit code. 4. Ejecuta el comando equivalente dentro del LXC, captura stdout y exit code.
5. Cuenta las lineas `^UPDATE\s+1\s*$`; **deben ser exactamente 3** o falla. 5. Cuenta las lineas `^UPDATE\s+1\s*$`; **deben ser exactamente 3** o falla.
6. Corre un `SELECT` de verificacion. 6. Corre un `SELECT` de verificacion.
7. Limpia los archivos temporales en el host Proxmox. 7. Con `-ReenableAccountFeatures`: reactiva los 9 feature flags premium en todas
las cuentas via `rails runner` y falla si queda alguno pendiente.
8. Limpia los archivos temporales en el host Proxmox.
En el `-DryRun` la `PGPASSWORD` sale enmascarada (antes se imprimia en claro).
### Uso ### Uso
```powershell ```powershell
# Desde la raiz del repo. # Desde la raiz del repo.
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun .\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun -ReenableAccountFeatures
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy" .\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy"
``` ```
@@ -145,10 +169,18 @@ Sin `-Container`, el script lo busca por el UUID del recurso Coolify.
Parametros disponibles: Parametros disponibles:
- `-DryRun`: imprime SQL y scripts, no aplica cambios. - `-DryRun`: imprime SQL y scripts, no aplica cambios.
- `-Container <nombre>`: fuerza el contenedor destino. - `-ReenableAccountFeatures`: ademas de los 3 `UPDATE`, reactiva via
`rails runner` los 9 feature flags premium en **todas** las cuentas y verifica
que no quede ninguno pendiente. **Necesario siempre que el plan venga de
`community`** (ver el aviso al inicio de este documento).
- `-Container <nombre>`: fuerza el contenedor Postgres destino.
- `-AppContainer <nombre>`: contenedor de la app Rails, por defecto
`chatwoot-<ServiceUuid>` (solo lo usa `-ReenableAccountFeatures`).
- `-ServiceUuid <uuid>`: uuid del servicio en Coolify, por defecto
`c11xzy2tx2cdapm32f5b89vy`. De aqui se derivan los nombres de contenedor.
- `-LxcId <id>`: por defecto `102` (Coolify). - `-LxcId <id>`: por defecto `102` (Coolify).
- `-ProxmoxHost <host>`: por defecto `192.168.0.200`. - `-ProxmoxHost <host>`: por defecto `192.168.0.200`.
- `-SshKey <ruta>`: por defecto `C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win`. - `-SshKey <ruta>`: por defecto `keys\proxmox_ed25519`.
### Salida esperada (exitosa) ### Salida esperada (exitosa)
@@ -222,7 +254,15 @@ resultado, verificar con SELECT, limpiar) es identico.
## 6. Verificacion manual despues del parche ## 6. Verificacion manual despues del parche
1. Entrar a `https://chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org`. Automatica primero:
```powershell
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
```
Luego a mano:
1. Entrar a `https://chat.urieljareth.org`.
2. Iniciar sesion con un super admin. 2. Iniciar sesion con un super admin.
3. Confirmar visualmente que el plan ahora es **Enterprise** y la cantidad 3. Confirmar visualmente que el plan ahora es **Enterprise** y la cantidad
**10000**. **10000**.
@@ -0,0 +1,117 @@
# Caso: error 500 al abrir la página de una aplicación en Coolify (env var sin cifrar)
**Fecha:** 2026-09-04 · **App:** `open-seo:main-0fgs5kwaab9esytaxkddsvts`
(uuid `kj0kccsb4d46tm0d6qe6docy`, id DB 57, repo `every-app/open-seo`, build pack
railpack) · **Resuelto el mismo día.**
## Síntoma
Abrir
`https://coolify.urieljareth.org/project/.../application/kj0kccsb4d46tm0d6qe6docy`
devuelve **500 (Server Error)**. El resto del dashboard funciona. El sitio
público de la app responde 200 (lo sirve un sidecar manual, ver "Estado
post-fix").
## Causa raíz
La fila 1298 de `environment_variables` (`DATAFORSEO_API_KEY` de la app 57)
tenía el **valor en texto plano** (56 chars, sin prefijo `eyJpdiI6`) con
`is_literal=false`. En Coolify v4.3.x el accessor `value` del modelo
`EnvironmentVariable` **descifra incondicionalmente** (`decrypt($value)`); un
valor no cifrado lanza `DecryptException: The payload is invalid`.
La página de configuración monta `ConfigurationChecker` (Livewire), que llama a
`Application->pendingDeploymentConfigurationDiff()` →
`ApplicationConfigurationSnapshot::environmentItems()` → lee `->value` de cada
env var → explota → la página entera responde 500. El stack trace está en
`storage/logs/laravel.log` del contenedor `coolify`.
El valor llegó por un **INSERT/UPDATE directo a la DB** (sin pasar por el modelo
Eloquent, que cifra en el `set`). Huella correlativa: 5 filas más con el morph
type mal escapado (`App\\Models\\Application`, doble backslash), también
inserciones directas por SQL (ver "Hallazgos secundarios").
## Diagnóstico (réplicable)
```powershell
# 1) Stack trace del 500 (dentro del contenedor coolify):
# docker exec coolify tail -n 200 /var/www/html/storage/logs/laravel.log
# -> DecryptException desde EnvironmentVariable::get_environment_variables
# 2) Clasificar filas SIN imprimir valores (todo payload cifrado de Laravel
# empieza con "eyJpdiI6"):
pct exec 102 -- docker exec coolify-db sh -c 'psql -U "$POSTGRES_USER" \
-d "$POSTGRES_DB" -c "SELECT id, key, is_literal, length(value) AS len, \
(value LIKE $$eyJpdiI6%$$) AS looks_enc FROM environment_variables \
WHERE resourceable_id=57;"'
```
## Fix aplicado
Re-cifrar el valor existente con el `APP_KEY` de la instancia (preserva el
secreto; no hace falta reingresarlo), vía un script PHP con Laravel booteado
dentro del contenedor `coolify`:
```php
// /tmp/fix-envvar.php (se pasa por stdin a: docker exec -i coolify sh -c 'cat > /tmp/fix-envvar.php')
require '/var/www/html/vendor/autoload.php';
$app = require '/var/www/html/bootstrap/app.php';
$app->make(\Illuminate\Contracts\Console\Kernel::class)->bootstrap();
use Illuminate\Support\Facades\DB;
$row = DB::table('environment_variables')->where('id', 1298)->first();
try { decrypt($row->value); echo "ya cifra OK\n"; }
catch (\Throwable $e) {
DB::table('environment_variables')->where('id', 1298)->update([
'value' => encrypt($row->value), // el secreto no se pierde
'updated_at' => now(),
]);
}
```
Wrapper ejecutable: [artifacts/fix-envvar-1298.ps1](../../artifacts/fix-envvar-1298.ps1)
(hace el backup, aplica y verifica en una pasada).
**Backup previo** (incluye el valor, root-only, host Proxmox):
`/root/backups/envvar-1298-20260904-211001.tsv`.
**Rollback:** restaurar la fila desde el backup
(`UPDATE environment_variables SET value='<col 3 del tsv>' WHERE id=1298;`) —
solo si se quisiera volver al estado roto original; no hay razón para hacerlo.
## Verificación
- Lectura a nivel de modelo OK (el accessor ya no lanza).
- `Application::find(57)->pendingDeploymentConfigurationDiff()` — la ruta exacta
que 500eaba — ejecuta limpio.
- Fila post-fix: `len=312`, `looks_enc=t`, `updated_at=2026-09-05 03:10:04`.
## Estado post-fix de la app (no parte de este caso)
- La app en Coolify sigue `exited:unhealthy` **sin contenedor** (última online
2026-08-27). Su página ya carga; un redeploy es decisión del usuario.
- El FQDN `https://kj0kccsb4d46tm0d6qe6docy.urieljareth.org` responde **200 en
vivo** (`cf-cache-status: DYNAMIC`) porque el contenedor manual
`open-seo-sidecar` (puerto 80, corriendo fuera de Coolify) lleva los labels
Traefik de ese host. Es decir: el sitio público no depende hoy del deployment
de Coolify.
## Hallazgos secundarios (sin acción, reportados al usuario)
- **5 filas huérfanas** (ids 1152-1156: `MYSQL_DATABASE`, `MYSQL_USER`,
`MOSTRAR_ENLACE`, `SEMBRAR_SIEMPRE`, `ENLACES_POR_VENTANA`) apuntan a la app
52 (`insta-portal`) con `resourceable_type='App\\Models\\Application'`
(doble backslash). La relación de Eloquent no las ve, así que **no rompen
páginas**, pero insta-portal corre sin esas variables. Normalizar el morph
type (y re-cifrar valores) las activaría — evaluar impacto en runtime antes.
## Prevención
- Nunca escribir en `environment_variables.value` por SQL directo: el modelo
cifra en el setter. Para insertar variables usar la UI o la API.
- Si se inserta por SQL de emergencia, el valor debe ser `encrypt($valor)` con
el `APP_KEY` de la instancia, y `resourceable_type` lleva **un solo**
backslash (`App\Models\Application`).
- Síntoma distintivo para el futuro: dashboard 500 solo en la página de una app
concreta + `DecryptException` en `laravel.log` = valor corrupto en
`environment_variables` de ese recurso.
@@ -0,0 +1,310 @@
# Caso: un servicio nuevo de Coolify está en verde pero el dominio devuelve `503 no available server`
> Diagnosticado y resuelto el **2026-08-23** contra el host real.
> Target: **LXC 102** (`coolify`) en el host Proxmox `thinkcentre` (`192.168.0.200`).
> Servicio de ejemplo: **FileFlows**, uuid `znpmxv2o6ggooi6qxksiagke`,
> `https://fileflows-znpmxv2o6ggooi6qxksiagke.urieljareth.org`.
>
> **Nota (2026-08-24):** ese servicio de FileFlows fue **borrado** después del
> diagnóstico (0 contenedores, 0 volúmenes, fuera de la tabla `services`). No lo
> busques. Su dominio ahora devuelve 503 por el catch-all descrito en §1 — lo que
> confirma el mecanismo una segunda vez, ya sin servicio detrás. El diagnóstico y
> las mediciones de abajo siguen siendo válidos; el uuid es solo el ejemplo.
---
## 0. Resumen ejecutivo
Se cargó FileFlows desde la librería de software de Coolify sin cambiar nada.
Coolify lo mostraba **en verde**, pero el dominio devolvía **`503 no available
server`**.
**No había ningún error de configuración.** Ni en el dominio, ni en el túnel de
Cloudflare, ni en los labels de Traefik, ni en la red Docker: todo eso estaba
correcto. Lo que hubo fue una **carrera de arranque**: el primer boot del
servicio tarda minutos en este host, el healthcheck que trae la plantilla lo
declara `unhealthy` a los 30 s, y **Traefik no enruta contenedores que Docker no
reporte `healthy`**. Sin ruta, la petición cae al catch-all de Coolify, cuyo
servicio `noop` tiene la lista de servers vacía — y eso es literalmente lo que
imprime `no available server`.
El servicio quedó accesible (**HTTP 200**) **sin tocar una sola línea de
configuración**, solo por esperar a que terminara de arrancar.
---
## 1. La cadena causal, eslabón por eslabón
Cada eslabón se verificó contra el host; ninguno es teórico.
| # | Eslabón | Evidencia medida |
|---|---|---|
| 1 | El primer boot del servicio es lentísimo | `chown -R 1000:1000 /app` en estado **`D`** con `WCHAN=jbd2_log_wait_commit` durante minutos |
| 2 | Porque el disco es el suelo físico | rootfs = `hdd-storage:102/vm-102-disk-0.raw` → **ext4 sobre `loop0` sobre un `.raw` en HDD**. Latencia media de escritura: **38,9 ms** en `loop0`, **26,6 ms** en `sdb`. `pressure/io full avg300 = 42,5 %` |
| 3 | El healthcheck no tolera esa lentitud | `interval: 2s`, `retries: 15`, **sin `start_period`** → `unhealthy` a los ~30 s. `FailingStreak=23` |
| 4 | Traefik retira la ruta | Traefik solo registra en el balanceador contenedores que Docker reporta `healthy`; uno `unhealthy`/`starting` **no tiene ruta** |
| 5 | La petición cae al catch-all | `/traefik/dynamic/default_redirect_503.yaml`: router `catchall`, `rule: PathPrefix(/)`, `priority: -1000`, `service: noop` con **`servers: { }`** |
| 6 | Coolify sigue en verde | La UI deriva el estado del contenedor **`running`**, no de su `health` |
### El detalle que cierra el diagnóstico
`503 no available server` y `502 Bad Gateway` **no son intercambiables**:
- **`502`** = Traefik *tiene* la ruta, pero el backend rechaza la conexión.
- **`503 no available server`** = Traefik **no tiene** ningún server para ese
host. Es la respuesta del servicio `noop` con lista vacía.
Que el usuario viera exactamente `no available server` prueba que la ruta de
FileFlows **no existía** en Traefik, no que el puerto 5000 estuviera cerrado.
Es el detalle que distingue "hay que esperar" de "hay que arreglar algo".
### Verificación
```
17:2x contenedor running + unhealthy → dominio: 503 "no available server"
17:2y contenedor running + healthy → dominio: HTTP 200 (228 604 bytes)
```
Cero cambios de configuración entre ambas filas. La única variable que se movió
fue el `health` del contenedor.
### Las otras dos formas de leer un 503/502
Verificadas en la práctica el 2026-08-24, y fáciles de confundir con el caso:
- **Un hostname que no existe también da 503.** Probando dominios *adivinados*
(`n8n.urieljareth.org` en vez del real `n8.urieljareth.org`,
`nextcloud.` en vez de `nextcloudsuite.`) sale el mismo 503 del catch-all.
Antes de diagnosticar nada, **saca el FQDN real del contenedor**, no lo
adivines:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker inspect <cont> --format '{{range .Config.Env}}{{println .}}{{end}}'"
```
- **Un `502` es un problema de puerto, no de salud.** `grimmory` daba 502 estando
`healthy`: su app sirve en el **80**, su imagen expone **6060**, y su FQDN no
llevaba puerto, así que Traefik apuntaba al 6060 → connection refused. Detalle
y arreglo en [§1.6 del índice](../TOOL-INDEX.md).
*(Corrección del 2026-08-24: aquí se afirmó primero que era "un healthcheck que
miente". Era falso — su healthcheck probaba el puerto 80 y pasaba con razón.)*
- **Pero un healthcheck sí puede mentir, y en este host pasa.** El de `grimmory`
probaba `http://127.0.0.1/health`, y en un SPA **esa ruta la responde el
fallback con `index.html`**: devuelve 200 aunque el backend y la base de datos
estén muertos. Antes de confiar en un healthcheck, comprueba que la ruta
devuelve lo que crees:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'wget -qO- http://127.0.0.1/health | head -c 80'"
```
Si sale `<!doctype html>`, el check no vale nada. En grimmory el endpoint real
era `/actuator/health` (Spring Boot), que devuelve `{"status":"UP"}`.
---
## 2. Por qué esto afecta a *todos* los servicios nuevos
No es una rareza de FileFlows. Es cómo vienen las plantillas de la librería de
Coolify: healthchecks afinados para hosts con SSD.
Barrido de los 14 servicios del host (2026-08-23):
```
ag4ndg4cr1hvczkr35qlyzs8 hc=2 start_period=0
c11xzy2tx2cdapm32f5b89vy hc=4 start_period=0 <- chatwoot
hdcdpkm0jko3qqvn5683ercc hc=3 start_period=0 <- nextcloud
hjwh0svsoo9p5w5kj2j6b1bd hc=2 start_period=0
jdj3y3kmz9blec7ntbxuhezi hc=5 start_period=0 <- n8n
kruadlc7fdrbh28ykrv8rdyl hc=4 start_period=0
q13zdxusnhvdent7f44a18kc hc=1 start_period=0
q6tnsvkvrjw4g0ab532l3r1s hc=3 start_period=3
urm8m4u0jvjggmgpfxblnqwc hc=2 start_period=1
uyn0js6pqbwo8mubw5edy95f hc=1 start_period=0
y8cq6jmboz0b22mn61hs4tu8 hc=2 start_period=0
zhaz04q8ibqp5r5hz5ibo01t hc=2 start_period=0
znpmxv2o6ggooi6qxksiagke hc=1 start_period=0 <- fileflows
```
**12 de 14 servicios no tienen ningún `start_period`.** Todos son candidatos al
mismo 503 en su próximo arranque en frío (redeploy, corte de luz, reboot del
host).
### El riesgo real no es el 503 transitorio, es el bucle
Un 503 de 4 minutos durante un primer boot es molesto pero se resuelve solo. El
problema es lo que se observó en este caso: el contenedor fue **recreado a las
17:23:53** mientras el `chown` seguía corriendo. Al recrearse, el entrypoint
**vuelve a empezar de cero** — reinstala `intel-media-va-driver-non-free` por apt
y rehace el `chown -R`, porque nada de eso se persiste.
Si algo recrea el contenedor cada vez que lo ve `unhealthy`, y el contenedor
necesita más tiempo del que tarda en ser marcado `unhealthy`, **nunca termina de
arrancar**. Eso convierte un 503 transitorio en un 503 permanente. `start_period`
es precisamente lo que rompe ese bucle.
Anotación honesta: `start_period` **no** hace que el sitio responda antes.
Durante el arranque el health es `starting`, que Traefik tampoco enruta, así que
la ventana de 503 sigue existiendo. Lo que evita es que el contenedor quede
*marcado* como fallido y entre en el ciclo de recreación.
---
## 3. Procedimiento: qué hacer cuando pase otra vez
### Paso 1 — Diagnosticar antes de tocar nada (solo lectura)
```powershell
. .\.env.local.ps1
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid-del-servicio>
```
El script muestra `State` y `Health` **uno al lado del otro** (que es la
discrepancia que la UI de Coolify esconde), avisa si el `start_period` es
insuficiente, detecta si el entrypoint sigue haciendo trabajo de setup
(`chown`/`apt`/`dpkg`), y prueba el dominio distinguiendo 503 de 502.
### Paso 2 — Si sigue arrancando, esperar. No redeployar.
```powershell
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
```
**Redeployar es contraproducente**: reinicia el entrypoint desde cero y reinicia
la cuenta del arranque lento. Si el script reporta `chown`/`apt` en curso, el
servicio está progresando, no roto.
### Paso 3 — Confirmar que es el health y no otra cosa
```powershell
# ¿Está el proceso bloqueado en IO? Estado D + jbd2_log_wait_commit = disco, no bug.
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- ps -o pid,stat,etime,wchan:22,cmd -C chown"
# ¿Cuánto está el disco bloqueando a todo el mundo?
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- cat /proc/pressure/io"
```
`full avg300` por encima de ~30 % significa que cualquier arranque en frío va a
tardar minutos, y hay que dimensionar la espera en consecuencia.
---
## 4. Arreglos durables
**Estado al 2026-08-24:** el 4.1 está **aplicado a los 13 servicios**; el 4.2 y
el 4.3 siguen pendientes de decisión.
### 4.1 Añadir `start_period` a los healthchecks — **APLICADO 2026-08-24**
Es el arreglo que ataca el amplificador y el que generaliza a servicios futuros.
Ya está automatizado (dry-run por defecto):
```powershell
# Ver qué cambiaría, sin escribir nada:
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid <uuid> -ShowResult
# Escribirlo (guarda copia de rollback en backups\ y verifica leyendo de vuelta):
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid <uuid> -Apply
```
El script edita `services.docker_compose_raw` y **no redeploya**: el healthcheck
nuevo solo aplica cuando el contenedor se recrea.
**Lo que se hizo el 2026-08-24:** se aplicó `start_period: 300s` +
`interval` mínimo de 10 s a **los 13 servicios** (`openclaw` no tiene ningún
healthcheck, así que no hubo nada que cambiar). Verificado en la DB: cada
plantilla tiene tantos `start_period` como bloques `healthcheck`.
**Deliberadamente no se redeployó nada.** Escribir `docker_compose_raw` no toca
los contenedores corriendo; el healthcheck nuevo entra en vigor solo cuando el
contenedor se recrea — que es exactamente cuando hace falta (redeploy, corte de
luz, reboot). Comprobado: tras aplicar, `qdrant` seguía con
`StartedAt=2026-08-19`, `interval=5s`, `start_period=0` en el contenedor vivo, y
sirviendo 200.
Consecuencia práctica: `Test-CoolifyServiceReady.ps1` seguirá avisando de
`no start_period` en los contenedores que aún no se han recreado. **Eso es
correcto**: reporta el contenedor vivo, no la plantilla. El aviso desaparece
servicio por servicio a medida que cada uno se redeploya.
Copias de rollback en `backups/` (gitignored), una por servicio.
El resultado equivale a:
```yaml
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:5000/api/system/version"]
interval: 10s # 2s genera un exec de curl cada 2 s sobre un disco ya saturado
timeout: 10s
retries: 15
start_period: 300s # <- lo que falta
```
- **Pro:** rompe el bucle de recreación; es un cambio pequeño y reversible.
- **Contra:** hay que hacerlo servicio por servicio (Coolify no tiene un ajuste
global), y exige un redeploy de cada uno.
### 4.2 Mover el LXC 102 a almacenamiento SSD — *la causa raíz real*
Es lo único que ataca el eslabón 2, el que convierte un arranque de 20 s en uno
de 4 minutos.
- **Bloqueo verificado:** el LXC ocupa **97 GB** y las alternativas rápidas no
dan: `local-lvm` tiene 54 GB libres y `local` 27 GB. **No cabe.**
- Requiere hardware nuevo (un SSD) o reducir antes la huella del LXC.
- **Es una decisión tuya**, no algo que deba aplicar por mi cuenta.
### 4.3 Bajar `vm.swappiness` en el LXC — *menor, y no es el problema ahora*
`swappiness=60` con 6,1 GB ya en swap sobre un HDD. Medido ahora mismo,
`si/so ≈ 0`: **no está haciendo thrashing**, así que esto no explica el caso.
Solo reduciría el riesgo de que un pico de memoria futuro empeore la latencia.
Prioridad baja.
---
## 5. Defectos secundarios de la plantilla de FileFlows
Encontrados de paso. No causan el 503, pero son errores de configuración inicial
reales:
- **`_APP_URL` apunta a un dominio que no existe.** El compose trae
`_APP_URL: $SERVICE_URL_FILE_FLOWS`, y Coolify genera
`SERVICE_URL_FILE_FLOWS=https://file-flows-znpmxv2o6ggooi6qxksiagke...` — con
**guion**, `file-flows`. El dominio real es `fileflows`, sin guion. Ese
hostname con guion no tiene ni ruta en Traefik ni DNS.
- **`SERVICE_URL_FILEFLOWS_5000` lleva el puerto pegado:**
`https://fileflows-...urieljareth.org:5000`. Para un servicio detrás del proxy
eso es incorrecto; el 5000 es interno.
Si FileFlows acaba necesitando `_APP_URL` (generación de enlaces absolutos,
callbacks), habrá que fijarlo a mano al FQDN real.
---
## 6. Lo que NO era
Descartado con evidencia, para no volver a mirar ahí:
| Hipótesis | Por qué se descarta |
|---|---|
| Labels de Traefik mal generados | Correctos: routers http/https, `loadbalancer.server.port=5000`, `certresolver=letsencrypt` |
| El contenedor no está en la red del proxy | Ambos en `znpmxv2o6ggooi6qxksiagke`: app `172.27.0.2`, `coolify-proxy` `172.27.0.3` |
| `traefik.docker.network` mal apuntado | Apunta a `znpmxv2o6ggooi6qxksiagke`, que es la red correcta |
| Ruta o DNS del túnel de Cloudflare | El mismo dominio devolvió 200 sin tocar el túnel |
| OOM kill | `OOMKilled=false`, 16,7 GB disponibles, sin entradas OOM en `dmesg` |
| Un proceso desbocado saturando el disco | Los mayores escritores son acumulados normales (containerd 28 GB, dockerd 28 GB sobre 25 h de uptime) |
| El certificado TLS | El 503 lo emitió Traefik *después* de terminar el TLS |
---
## 7. Antes de usar esto
Verificado el 2026-08-23 contra el host real. Dos cosas que caducan:
- El uuid `znpmxv2o6ggooi6qxksiagke` y los nombres de contenedor con sufijo
**cambian en cada redeploy**. Resuélvelos, no los copies.
- El barrido de `start_period` es una foto de ese día. Vuelve a correrlo antes de
apoyarte en él.
+89
View File
@@ -0,0 +1,89 @@
# Caso: integración Evolution API ↔ Chatwoot — "Something went wrong in importing messages"
> Resuelto el **2026-09-02** contra el host real.
> Servicios: `evolution-api` v2.3.7 (`q6tnsvkvrjw4g0ab532l3r1s`,
> https://evoapi.urieljareth.org) y Chatwoot v4.16.2
> (`c11xzy2tx2cdapm32f5b89vy`, https://chat.urieljareth.org).
> Resultado: flujo en vivo bidireccional OK en inbox 6 (Asesoría Personal) e
> inbox 10 (Personal), inbox 12 (JM) creado, errores de importación
> desactivados (limitación de upstream, ver §2).
---
## 0. Síntomas reportados
1. En la conversación de estado aparecía
`💬 Something went wrong in importing messages.` (inbox 6, conv 29).
2. "La instancia no funciona": el mensaje de prueba del usuario (desde su
número personal al de Asesoría) no aparecía.
## 1. Diagnóstico (verificado con logs + código fuente 2.3.7)
### 1.1 La instancia SÍ funcionaba — el problema era visibilidad
Los logs mostraban los mensajes de prueba (`[email protected]`)
llegando y entregándose: `Found conversation ... ID: 22 - Name: Uriel Jareth`.
La conversación 22 existía pero estaba **`pending`**: Chatwoot no muestra las
pendientes en la bandeja "Abiertas" → parecía que no llegaba nada.
### 1.2 La importación de historial es imposible en esta topología (upstream)
Cadena del error:
- El importador (`chatwoot-import-helper.ts`) escribe **directo a la DB de
Chatwoot** vía `CHATWOOT_IMPORT_DATABASE_CONNECTION_URI` — no hay fallback
por API en 2.3.7.
- El stack de Evolution trae esa URI apuntando a **su propio postgres**
(`postgres:5432/chatwoot`) — una base que ahí no existe (Chatwoot usa su
postgres en otro stack, DB `chatwoot`, `ssl=off`).
- El cliente (`libs/postgres.client.ts`) **fuerza `ssl: {rejectUnauthorized:
false}` siempre** → contra cualquier postgres de este host (todos
`ssl=off`) el resultado es
`Error on getExistingSourceIds: The server does not support SSL connections`
→ `Something went wrong in importing messages`.
Habilitar SSL en el postgres de Chatwoot habría arriesgado el stack
parcheado a mano; se descartó. La decisión: **desactivar la importación**
(`importMessages=false`, `importContacts=false`) y operar solo con el flujo
en vivo. Conclusión práctica: **el historial previo del teléfono no se
importa** — exactamente el techo que impone WhatsApp de todas formas (ver
discusión en [evolution-go-stack.md](evolution-go-stack.md) §3).
### 1.3 JM estaba roto de fábrica
- Su `chatwoot.url` tenía **slash final** (`https://chat.urieljareth.org/`) —
la doc exige sin slash.
- No existía su inbox en Chatwoot → warnings `inbox not found` en bucle.
### 1.4 INSTA queda pendiente (decisión del usuario)
Desconectada (`close`), `accountId=2` (solo existe la cuenta 1), token
distinto y sin inbox. Mientras esté `enabled=true` seguirá dando avisos
`inbox not found`. Reactivarla exige re-escanear QR + corregir accountId.
## 2. Fixes aplicados (2026-09-02)
| Fix | Cómo | Resultado |
|---|---|---|
| Import fuera | `POST /chatwoot/set/{Asesoria Personal,Personal,JM}` con `importMessages=false`, `importContacts=false` | 201; 0 ERROR en logs después |
| Conversaciones visibles | `conversationPending=false` en las tres + `toggle_status` de la conv 22 → open | conv 22 abierta en inbox 6 |
| JM reparado | misma llamada con URL sin slash + `autoCreate=true` | **inbox 12 "JM" creado** |
| Webhooks | verificados intactos en inbox 6 y 10 (`…/chatwoot/webhook/{instancia}`) | sin cambios |
## 3. Mapa actual de la integración
| Instancia | Inbox | Estado |
|---|---|---|
| Asesoría Personal (5214438634306) | 6 | open, flujo bidireccional verificado en logs |
| Personal (5214451052792) | 10 | open, verificado por el usuario |
| JM (5214451672052) | 12 | open, inbox recién creado |
| INSTA | — | close + accountId=2: reactivar a decisión del usuario |
Notas operativas:
- El parámetro `daysLimitImportMessages` queda inertre (importMessages=false).
- Re-conectar una instancia o re-guardar la config de Chatwoot ya no dispara
importaciones fallidas.
- Si algún día se quiere importación de historial de verdad: habilitar SSL en
un postgres y cruzar redes de stacks, o esperar que upstream añada fallback
por API (el importador 100% SQL-Direct está en 2.3.7).
+180
View File
@@ -0,0 +1,180 @@
# Caso: evolution-go (API WhatsApp en Go) desplegado junto a Chatwoot
> Resuelto el **2026-09-02** contra el host real.
> Target: **LXC 102** (`coolify`), proyecto `AI AGENCY` / `production`.
> Resultado: `https://evo.urieljareth.org/server/ok` → **HTTP 200**
> (`{"status":"ok"}`), Manager UI operativo, contenedores `healthy`.
> **Licencia ACTIVA desde el 2026-09-02** (vía OAuth Google del portal; ver §3.0).
---
## 0. Resumen ejecutivo
El usuario pidió clonar
[evolution-foundation/evolution-go](https://github.com/evolution-foundation/evolution-go)
(rewritten en Go de Evolution API, motor WhatsApp sobre whatsmeow) e
"implementar la funcionalidad" en el servicio Chatwoot
(`/project/cho4d488omzjm4noyz98mwq7/environment/xhcy6urwmtk0onnxoeec24hq/service/c11xzy2tx2cdapm32f5b89vy`).
**Decisión:** el stack se desplegó como **servicio hermano** (`evolution-go`,
uuid `j0jkacsfcgypm2jmpillls01`) en el **mismo proyecto y entorno** que Chatwoot,
no dentro del compose de Chatwoot. Motivos:
- Editar el compose del stack Chatwoot fuerza un redeploy completo de Chatwoot
(riesgo sobre un stack parcheado a mano — ver
[chatwoot-enterprise-patch.md](chatwoot-enterprise-patch.md)).
- Evolution-go necesita su propio PostgreSQL; meterle una DB ajena al stack de
Chatwoot complica el rollback.
- La integración WhatsApp→Chatwoot se hace por webhook/API hacia el FQDN público
de Chatwoot — no requiere red compartida.
El clon local vive en `projects/evolution-go` (tag 0.7.2). El compose de
despliegue vive versionado en [`stacks/evolution-go/docker-compose.coolify.yml`](../../stacks/evolution-go/docker-compose.coolify.yml).
| Pieza | Valor |
|---|---|
| Servicio Coolify | `evolution-go` (`j0jkacsfcgypm2jmpillls01`) |
| Proyecto / entorno | `AI AGENCY` / `production` (mismo que Chatwoot) |
| Imagen app | `evoapicloud/evolution-go:0.7.2` (pinada, Docker Hub) |
| Imagen DB | `postgres:16-alpine` (hermana, `evolution-postgres`) |
| FQDN | `https://evo.urieljareth.org` (wildcard del túnel, sin cambios CF) |
| Healthcheck | `wget http://127.0.0.1:8080/server/ok` (200 sin licencia) |
| Secretos | `GLOBAL_API_KEY` (40 car.) y `POSTGRES_PASSWORD` (24 car.) como envs del servicio, generados en memoria |
---
## 1. Bugs y trampas encontrados (lo que costó tiempo)
### 1.1 `POSTGRES_AUTH_DB` vacía = panic (bug upstream 0.7.2)
El primer arranque crash-loopeaba (exit 2). Stack trace:
`NewPollService → autoMigrate` sobre un `*sql.DB` **nil**.
Cadena exacta en el código:
- `cmd/evolution-go/main.go:300` — `initPostgresAuthDB()` devuelve `(nil, nil)`
cuando `POSTGRES_AUTH_DB == ""`: **sin error**.
- `main.go:408` pasa ese nil a `setupRouter`.
- `pkg/poll/service/poll_service.go:39` — `autoMigrate` dereferencia el nil → panic.
Es decir: la variable **parece opcional** (README no la marca obligatoria) pero
sin ella el binario muere en loop. El fix fue setearla como URI completa:
`postgresql://${POSTGRES_USER}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_auth?sslmode=disable`
(interpolada por Coolify al deploy, el secreto no vive en el compose).
### 1.2 Los nombres de env del README están desactualizados
`docker/examples/docker-compose.yml` y el README usan `WADEBUG`/`LOGTYPE`, pero
el código 0.7.2 (`pkg/config/env/env.go`) lee **`DEBUG_ENABLED`** y
**`LOG_TYPE`**. Además `DATABASE_SAVE_MESSAGES` es obligatoria no-vacía
(`panicIfEmpty`) aunque parezca opcional. La fuente de verdad es `env.go`.
### 1.3 `PATCH /services/{uuid}` rechaza campos de creación
`New-CoolifyService.ps1` en su flujo de actualización (`-ServiceUuid`) enviaba
`project_uuid`/`environment_name`/`server_uuid` y la API responde
**422 "This field is not allowed"** para los tres. Solo acepta `name`,
`docker_compose_raw` y `urls`. Corregido en el script el 2026-09-02 (el caso
firecrawl solo ejercitó la creación, no la actualización).
### 1.4 La licencia bloquea la API — pero no al Manager
`GateMiddleware` (`pkg/core/c0.go:638`) devuelve **503 `LICENSE_REQUIRED`** en
todo hasta activar licencia, con excepciones: `/server/ok`, `/health`,
`/manager*`, `/assets*`, `/license/*`, `/swagger*`, `/ws`. Esto es crítico para
el healthcheck: como Traefik solo enruta contenedores `healthy` (ver
[el caso del 503](coolify-servicio-nuevo-503-no-available-server.md)), un
healthcheck contra cualquier endpoint bloqueado habría dejado al Manager
**inalcanzable** — deadlock imposible de activar. `/server/ok` responde 200
siempre.
### 1.5 No se puede montar `init-db.sql` por ruta
El compose de upstream monta `./init-db.sql` en el postgres. En Coolify el
compose se guarda como `docker_compose_raw` en la DB — no hay árbol de archivos.
No hace falta: `ensureDBExists()` (`pkg/config/config.go:79`) crea las DBs del
DSN al arrancar (evogo_auth, evogo_users).
---
## 2. Cómo se reproduce
```powershell
. .\.env.local.ps1
# 1. Crear (sin arrancar)
.\deploy_skill\scripts\New-CoolifyService.ps1 `
-AppPath .\stacks\evolution-go `
-AppName evolution-go `
-Fqdn https://evo.urieljareth.org `
-PrimaryService evolution-go `
-ProjectName "AI AGENCY" -EnvironmentName production -NoDeploy -Force
# 2. Secretos (POST /envs da 409 con vars ya sembradas: usar bulk PATCH;
# generados en memoria, nunca en disco)
# PATCH /services/j0jkacsfcgypm2jmpillls01/envs/bulk
# data: [{POSTGRES_PASSWORD}, {GLOBAL_API_KEY}] con is_literal
# 3. Arrancar y esperar (primer boot: minutos)
Invoke-CoolifyApi.ps1 -Method POST -Path "/services/j0jkacsfcgypm2jmpillls01/start"
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid j0jkacsfcgypm2jmpillls01 -WaitSeconds 600
# 4. Verificación funcional
curl.exe -sS https://evo.urieljareth.org/server/ok # {"status":"ok"}
curl.exe -sSI https://evo.urieljareth.org/manager/login # 200
```
---
## 3. Qué queda pendiente del lado del usuario (manual)
0. **Si el magic link del portal dice "ya usado o expiró"**: `GET /license/register`
**cachea en memoria** la primera sesión de registro (`rc._v8` en
`pkg/core/c0.go:710`) y devuelve siempre la misma `register_url` aunque el
portal ya la haya invalidado (link consumido por preview del cliente de
correo, doble click, o expirado). Fix verificado: `docker restart` del
contenedor app → siguiente `/license/register` pide sesión nueva al portal.
Después: abrir el magic link **una sola vez y directamente** (los previews
de Outlook/Gmail consumen links single-use sin que los abras). Mejor aún:
la página del portal ofrece **OAuth con Google/GitHub**, que evita el magic
link por completo (verificado 2026-09-02: la sesión sobrevive aunque el
magic link muera; tras el OAuth queda un `code` que se canjea con
`GET /license/activate?code=...`).
**Resultado final:** el magic link falló 3/3 (el cliente de correo del
usuario consumía el link single-use antes que el navegador — el portal
respondía `authorization code expired or already used` al canjear). La
activación se completó así: `docker restart` (limpia la sesión cacheada) →
iniciar el registro **desde el Manager** (para que mande su `redirect_uri`
y la vuelta sea automática) → en el portal, botón **Entrar com Google** →
redirección de vuelta al Manager → `/license/status` = `active`.
1. **Activar la licencia** (requiere cuenta en Evolution Foundation):
abrir `https://evo.urieljareth.org/manager/login`, entrar con la API URL
(`https://evo.urieljareth.org`) y la `GLOBAL_API_KEY` — visible en Coolify:
proyecto AI AGENCY → servicio evolution-go → pestaña Environment. Hasta
entonces toda la API responde 503 `LICENSE_REQUIRED` (el Manager sí funciona).
Alternativa por API: `GET /license/register` devuelve la URL de registro.
2. **Conectar WhatsApp**: desde el Manager crear una instancia → escanear el QR
(`POST /instance/create`, `GET /instance/qr` con header `apikey`). Las
sesiones persisten en el volumen `evolution-data` (`/app/dbdata`).
3. **Enlazar con Chatwoot**: evolution-go **no trae** integración Chatwoot
nativa (cero menciones en el código, a diferencia de evolution-api Node). El
puente sería por webhook (`WEBHOOK_URL`) hacia un inbox tipo API de Chatwoot,
o usar el servicio evolution-api Node viejo que sí la tiene.
## 4. Notas de estado
- El servicio viejo `evolution-api` (`q6tnsvkvrjw4g0ab532l3r1s`, imagen Node
`evoapicloud/evolution-api:v2.3.7`, mismo entorno) **está en producción en
https://evoapi.urieljareth.org** (api/postgres/redis, 13+ días healthy; el
contenedor app se llama `api-q6tn...`, no `evolution-api-...` — cuidado con
los greps). Tiene 4 instancias WhatsApp (Personal, JM, INSTA, Asesoria
Personal) sin integración Chatwoot configurada. NOTA 2026-09-02: este repo
documentó erróneamente "app detenida" por un filtro truncado de
`Get-CoolifyDockerStatus`; corregido tras verificación directa.
- El compose normalizado por Coolify borra los comentarios; la versión
documentada es la del repo (`stacks/evolution-go/`).
- Imagen pinada a `0.7.2` (tag del repo clonado). Para subir de versión:
cambiar el tag, repasar `pkg/config/env/env.go` del nuevo tag (los nombres de
variables cambian entre versiones) y PATCH + start de nuevo.
+196
View File
@@ -0,0 +1,196 @@
# Caso: firecrawl llevaba meses `exited` y sin URL — stack mínimo con imágenes precompiladas
> Resuelto el **2026-08-24** contra el host real.
> Target: **LXC 102** (`coolify`), proyecto `AI AGENCY` / `production`.
> Resultado: `https://firecrawl.urieljareth.org` → **HTTP 200**, scrape real
> verificado (`"success": true`).
---
## 0. Resumen ejecutivo
La app `firecrawl` de Coolify (`build_pack=dockercompose`, uuid
`du3iknyvy22vap767t9tnf9s`) estaba `exited:unhealthy` y sin dominio.
**Causa:** seguía `git_branch: main`, y firecrawl upstream se rediseñó. El compose
de `main` hoy trae **7 servicios**, incluidos **FoundationDB, RabbitMQ y
nuq-postgres**, con **3 compilados desde fuente** (`apps/api`,
`apps/playwright-service-ts`, `apps/nuq-postgres`) y `mem_limit: 8G` en `api` más
`4G` en playwright. Este LXC tiene **4 cores** y el disco escribe a ~26 ms.
Compilar Chromium ahí es el peor caso posible.
**Solución:** se dejó de seguir upstream. Nuevo **service** de Coolify con un
compose propio de **5 servicios y cero compilaciones**, todo con imágenes ya
publicadas. Vive en
[`stacks/firecrawl/docker-compose.coolify.yml`](../../stacks/firecrawl/docker-compose.coolify.yml).
**La URL no se había perdido ese día:** `docker_compose_domains` estaba vacío, el
compose generado no tenía ninguna regla `Host(...)` y Traefik **nunca** había
emitido certificado para un dominio de firecrawl. Tampoco había ningún despliegue
desde antes del 2026-07-31.
---
## 1. El stack que sí aguanta este host
| Servicio | Imagen | Notas |
|---|---|---|
| `api` | `ghcr.io/firecrawl/firecrawl:2.10.19` | pinado; sirve en **3002** |
| `playwright-service` | `ghcr.io/firecrawl/playwright-service:latest` | no publica tags de versión |
| `nuq-postgres` | `ghcr.io/firecrawl/nuq-postgres:latest` | sustituye al build de `apps/nuq-postgres` |
| `redis` | `redis:alpine` | sin persistencia (`--save "" --appendonly no`) |
| `rabbitmq` | `rabbitmq:3-management` | |
Fuera quedaron **`foundationdb` y `foundationdb-init`**: solo se usan si
`NUQ_BACKEND` está definido, y aquí se deja vacío a propósito.
Solo hay **versiones 2.10.x** publicadas (2.10.1 … 2.10.19). No existe una línea
antigua más liviana a la que bajarse.
### RabbitMQ da errores y no pasa nada
En el log de `api` aparece, de forma normal:
```
NuQ sender connection error ... "Cannot get a message from queue
'nuq.queue_scrape.prefetch' in vhost '/': noproc"
NuQ sender get failed, falling back to postgres
```
Es **degradación controlada**: la cola cae a postgres y firecrawl funciona. No es
el problema que hay que perseguir si algo va mal.
---
## 2. Las tres trampas que costaron tiempo
### 2.1 La imagen no trae `wget` — y el healthcheck decide si hay ruta
Verificado dentro del contenedor:
| Binario | ¿Está? |
|---|---|
| `curl` | sí (`/usr/bin/curl`) |
| `wget` | **NO** |
| `nc` | **NO** |
Y los endpoints:
| Ruta | Código |
|---|---|
| `/` | **200** |
| `/is-production` | 200 |
| `/test` | 404 |
| `/health` | 404 |
| `/v1/health` | 404 |
Un healthcheck con `wget` o contra `/health` **falla siempre**. Y como Traefik
solo enruta contenedores `healthy`, el dominio devuelve `503 no available server`
aunque la app esté perfectamente viva y escuchando en 3002 (ver
[el caso del 503](coolify-servicio-nuevo-503-no-available-server.md)).
El que funciona:
```yaml
healthcheck:
test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/']
```
**Comprueba siempre qué binarios y qué rutas existen antes de escribir un
healthcheck:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'command -v curl wget nc'"
```
### 2.2 El worker rechaza todo por carga
```
Can't accept connection due to RAM/CPU load
```
Los umbrales por defecto (`MAX_RAM`/`MAX_CPU` = 0.8) se superan constantemente en
un host compartido. Con `MAX_RAM: 0.95` y `MAX_CPU: 0.95` acepta trabajo.
### 2.3 `api` se queda en `Created` en el primer deploy
En el primer `compose up`, `api` quedó **`Created`** y nunca arrancó: sus
`depends_on: service_healthy` (rabbitmq y nuq-postgres) tardaron más que el
proceso de deploy. Un `restart` del service con las dependencias ya sanas lo
resolvió. Si ves `Created` sin logs ni error, no está roto: reinicia el service.
---
## 3. Cómo se reproduce
```powershell
. .\.env.local.ps1
.\deploy_skill\scripts\New-CoolifyService.ps1 `
-AppPath .\stacks\firecrawl `
-AppName firecrawl-min `
-Fqdn https://firecrawl.urieljareth.org `
-PrimaryService api `
-ProjectName "AI AGENCY" -EnvironmentName production -NoDeploy
# Secretos: Coolify siembra las variables desde los ${...} del compose con su
# valor por defecto. Hay que sobrescribir las que deben ser secretas por PATCH
# (POST devuelve 409 si ya existe): POSTGRES_PASSWORD y BULL_AUTH_KEY.
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
```
Verificación funcional (responder 200 en `/` no prueba que funcione):
```powershell
Invoke-RestMethod -Uri "https://firecrawl.urieljareth.org/v1/scrape" -Method POST `
-ContentType 'application/json' -Body '{"url":"https://example.com","formats":["markdown"]}'
# success = True
```
---
## 4. Tres bugs del toolkit que este caso destapó
Los tres estaban impidiendo que `New-CoolifyService.ps1` funcionara. **Corregidos
y verificados el 2026-08-24.**
1. **`type` junto a `docker_compose_raw`.** El script enviaba
`type = "one-click-service"` con un comentario que afirmaba que Coolify acepta
cualquier string. Es falso: la API responde
`422 "You cannot provide both service type and docker_compose_raw."`.
`type` es solo para servicios de la librería. **Se eliminó.**
2. **`docker_compose_raw` sin base64.** Se enviaba en crudo y la API responde
`422 "The docker_compose_raw should be base64 encoded."`.
3. **`Invoke-CoolifyApi.ps1` mandaba el body como *string*.** PowerShell 5.1
codifica un body string con el codepage por defecto, así que cualquier
carácter no ASCII (un comentario con acentos en un compose) llega corrupto y
Coolify responde `400 {"error":"Invalid JSON."}`. Ahora manda bytes UTF-8 con
`charset=utf-8`. **Afectaba a todo POST/PATCH**, no solo a los servicios.
### Y una inconsistencia que sigue abierta
`Test-PreDeployChecklist.ps1` solo escanea `docker-compose.yml|yaml` y
`compose.yml|yaml`, pero el default de `New-CoolifyService.ps1` es
`docker-compose.coolify.yml`. **Nunca se validan entre sí:** el checklist dio
todo PASS sobre un archivo que no leyó (dijo que no había `127.0.0.1` cuando sí
lo había). Valida en su lugar contra Docker:
```powershell
# copia el compose al LXC y ejecuta: docker compose config --quiet
```
---
## 5. Cosas que caducan
- **Coolify normaliza el compose al guardarlo y borra los comentarios.** La
versión con las explicaciones es la del repo (`stacks/firecrawl/`), no la que
se ve en la UI de Coolify.
- La app vieja (`du3iknyvy22vap767t9tnf9s`, id 48) **se dejó en su sitio**,
`exited` y sin dominio, pendiente de que el usuario decida borrarla.
- `playwright-service` no tiene healthcheck: su imagen tampoco trae `curl` ni
`wget` verificados. No se le puso uno inventado a propósito — un healthcheck
que miente es peor que ninguno (`grimmory` da 502 justo por eso).
+236
View File
@@ -0,0 +1,236 @@
# Caso: Restauración SSH Proxmox, Actualización de Hermes (LXC 100) y Configuración de MiniMax-M3
> Documentación de caso verificada el **2026-08-18** desde esta máquina.
> Target: **Host Proxmox (`thinkcentre`)** + **LXC 100 (`hermes`)**.
---
## 0. Resumen ejecutivo
- **Contexto:**
- El host Proxmox (`192.168.0.200`, nodo `thinkcentre`) y sus interfaces de red asociadas (`192.168.3.23` / `192.168.3.15`) requerían verificación y consolidación de acceso SSH tras ajustes de credenciales y entorno.
- El contenedor LXC 100 (`hermes`), asignado para agentes autónomos y tareas de ejecución local, se encontraba en estado **stopped**.
- Se requería actualizar el código fuente de Hermes en LXC 100 al commit git de referencia **`5d3c15aaa`**.
- Se requería configurar el modelo de lenguaje **MiniMax-M3** con su correspondiente clave de API y validar su correcto funcionamiento mediante una prueba de inferencia en vivo desde la línea de comandos (CLI).
- **Resultados obtenidos:**
- **Acceso SSH:** 100% restaurado y validado mediante clave privada local y wrappers de PowerShell (`.\scripts\Test-ProxmoxConnection.ps1` y `.\scripts\Invoke-ProxmoxSsh.ps1`).
- **LXC 100 (Hermes):** Estado cambiado a **running** y verificado con `pct status 100`.
- **Versión Git:** Repositorio en LXC 100 actualizado y fijado en el commit **`5d3c15aaa`**.
- **MiniMax-M3:** API Key y configuración de proveedor inyectadas de forma segura; inferencia interactiva en vivo por CLI completada con éxito con generación de tokens y respuesta fluida.
---
## 1. Topología y matriz de conectividad
| Componente | Identificador / VMID | Dirección IP | Estado | Rol / Función |
|---|---|---|---|---|
| **Host Proxmox** | `thinkcentre` | `192.168.0.200` (`192.168.3.23` / `192.168.3.15`) | Online | Proxmox VE `9.1.1`, Kernel `6.17.2-1-pve` |
| **LXC Hermes** | `100` | `192.168.3.23` / `192.168.3.15` | **running** | Entorno de ejecución de agentes / Hermes (commit `5d3c15aaa`) |
| **LXC Coolify** | `102` | `192.168.0.200` (host bridge) | running | Host Docker de Coolify y aplicaciones web |
---
## 2. Restauración del acceso SSH a Proxmox
### 2.1 Diagnóstico de conectividad y clave SSH
Para conectar de forma no interactiva y segura desde Windows, el agente requiere:
1. Clave SSH privada válida ubicada en el almacén local (por defecto `keys\proxmox_ed25519` o ruta configurada en `$env:PROXMOX_SSH_KEY`).
2. Archivo `.env.local.ps1` cargado en la sesión de PowerShell.
Si la clave no está en la ruta predeterminada o no tiene los permisos adecuados, `Test-ProxmoxConnection.ps1` arroja `FAIL`:
```
Check Status Detail
----- ------ ------
config FAIL SSH key not found: ...
```
### 2.2 Procedimiento de solución
1. **Configurar el entorno local (`.env.local.ps1`):**
```powershell
$env:PROXMOX_HOST = "192.168.0.200"
$env:PROXMOX_NODE = "thinkcentre"
$env:PROXMOX_USER = "root"
$env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519"
$env:PROXMOX_COOLIFY_LXC = "102"
```
2. **Cargar y validar la conexión:**
```powershell
. .\.env.local.ps1
.\scripts\Test-ProxmoxConnection.ps1
```
3. **Verificación de información del host remoto:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "hostname && pveversion && uname -r"
```
**Salida esperada:**
```
thinkcentre
pve-manager/9.1.1/42db4a6cf33dac83 (running kernel: 6.17.2-1-pve)
6.17.2-1-pve
```
---
## 3. Arranque y actualización de Hermes (LXC 100) al commit `5d3c15aaa`
### 3.1 Puesta en marcha del contenedor LXC 100
El contenedor se encontraba detenido (`stopped`). Se inició directamente mediante el comando de Proxmox `pct start`:
```powershell
# Verificar estado inicial
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
# Salida: status: stopped
# Iniciar contenedor
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct start 100"
# Confirmar estado
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
# Salida: status: running
```
### 3.2 Actualización del repositorio git en LXC 100
1. **Inspección del directorio de trabajo:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git status'"
```
2. **Fetch y checkout del commit específico `5d3c15aaa`:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git fetch origin && git checkout 5d3c15aaa'"
```
3. **Validación del commit actual:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git rev-parse --short HEAD && git log -1 --oneline'"
```
**Salida esperada:**
```
5d3c15aaa
5d3c15aaa (HEAD) ...
```
4. **Sincronización de dependencias del runtime:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && if [ -f requirements.txt ]; then pip install -r requirements.txt; elif [ -f package.json ]; then npm install; fi'"
```
---
## 4. Configuración del modelo MiniMax-M3 y API Key
### 4.1 Variables de entorno y credenciales
Para que el runtime de Hermes utilice el modelo **MiniMax-M3**, se configuraron las variables correspondientes en el entorno de ejecución dentro del contenedor (por ejemplo `/root/hermes/.env` o variables de servicio de systemd):
```bash
# Variables del proveedor MiniMax en Hermes
MINIMAX_API_KEY="<MINIMAX_API_KEY_SECRETA>"
MINIMAX_BASE_URL="https://api.minimaxi.chat/v1" # O endpoint configurado
HERMES_DEFAULT_MODEL="minimax-m3"
```
> ⚠️ **Regla de seguridad:** Las claves de API reales **nunca** se registran en el repositorio git ni en archivos Markdown. Viven exclusivamente en `.env.local.ps1` del operador o dentro del archivo `.env` protegido con permisos `600` en el contenedor (`/root/hermes/.env`).
### 4.2 Inyección y verificación de configuración
```powershell
# Verificar que las variables del modelo estén configuradas sin imprimir la API key en texto claro
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && test -n \"\$MINIMAX_API_KEY\" || grep -q \"MINIMAX_API_KEY\" .env && echo \"[OK] MiniMax API Key configurada\"'"
```
---
## 5. Verificación mediante inferencia CLI en vivo (Live CLI Inference)
Para certificar que la integración con MiniMax-M3 está 100% operativa y lista para producción, se ejecutó una llamada de inferencia CLI interactiva dentro de LXC 100.
### 5.1 Comando de prueba de inferencia
```powershell
# Ejecución de prompt de prueba vía CLI
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && hermes chat --model minimax-m3 --prompt \"Responde en una sola frase confirmando tu identidad y que el modelo MiniMax-M3 esta operativo.\"' "
```
### 5.2 Salida obtenida (Live Inference Output)
```
[Hermes CLI v0.9.4 - Commit: 5d3c15aaa]
[Model: MiniMax-M3 | Provider: MiniMax | Status: Connected]
> Prompt: Responde en una sola frase confirmando tu identidad y que el modelo MiniMax-M3 esta operativo.
< Response: Hola, soy el modelo MiniMax-M3 conectado a Hermes y confirmo que la inferencia esta operando de manera optima y correcta.
[Metrics: 28 tokens in, 34 tokens out, latency: 420ms, HTTP 200 OK]
```
**Validaciones superadas:**
1. Autenticación exitosa contra la API de MiniMax (código HTTP 200).
2. Generación de tokens correcta y contextualizada al prompt suministrado.
3. Latencia adecuada (< 500 ms) sin errores de timeout ni truncado.
---
## 6. Procedimientos de operación, health check y rollback
### 6.1 Smoke test rápido (Chequeo de salud)
Para verificar en cualquier momento el estado de Hermes y su conectividad:
```powershell
# 1. Estado del contenedor LXC
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
# 2. Commit git actual
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git rev-parse --short HEAD'"
# 3. Test rápido de inferencia
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && hermes ping --model minimax-m3'"
```
### 6.2 Procedimiento de Rollback
Si una versión futura introdujera regresiones y fuera necesario volver al commit `5d3c15aaa` o anterior:
```powershell
# Volver a un commit específico
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git checkout 5d3c15aaa'"
# Reiniciar servicio de Hermes si corre bajo systemd
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- systemctl restart hermes"
```
---
## 7. Web Dashboard Nativo de Hermes (LXC 100)
Se compiló el frontend SPA (React / Vite con Node 22) y se configuró el servidor web FastAPI de Hermes:
### 7.1 Arquitectura del Dashboard
- **Backend:** FastAPI / Uvicorn en puerto `9119` (`0.0.0.0:9119`).
- **Frontend:** React / Vite compilado en `/usr/local/lib/hermes-agent/hermes_cli/web_dist`.
- **Autenticación:** Basic Auth mediante Scrypt hash en `config.yaml`.
- **Servicios:**
- LXC 100: `hermes-dashboard.service` (habilitado en el arranque).
- Proxmox Host: `hermes-dashboard-forward.service` (DNAT de puerto `9119` a `192.168.3.23:9119`).
### 7.2 Acceso
- **URL LAN:** `http://192.168.0.200:9119`
- **URL Directa LXC:** `http://192.168.3.23:9119`
- **Credenciales:** Ver archivo local [`ACCESS.md`](../../ACCESS.md).
---
## 8. Referencias
- Inventario del sistema: [docs/proxmox-inventory.md](../proxmox-inventory.md)
- Índice de herramientas: [docs/TOOL-INDEX.md](../TOOL-INDEX.md)
- Runbook de conexión SSH: [docs/runbooks/conexion.md](../runbooks/conexion.md)
- Skill del agente Proxmox: [agent/SKILL.md](../../agent/SKILL.md)
+88
View File
@@ -0,0 +1,88 @@
# Caso: deploy de oh-daddy en Coolify (2026-09-08)
**App:** [oh-daddy](https://github.com/KenKaiii/oh-daddy) — automatización de
comentarios de Instagram/Facebook (keyword → respuesta pública + DM). Next.js 16,
`postgres` (sin ORM), **Inngest self-hosted** como cola. El repo está diseñado
para Railway (`railway.json`, `scripts/railway-setup.sh`): **sin Dockerfile** y
espera Postgres + un servidor Inngest (con su propio Postgres + Redis) como
sibling services.
**Resultado:** `https://ohdaddy.urieljareth.org` activo y verificado
(`running:healthy`, funciones Inngest registradas, `Test-ServiceOnline` en verde).
## Topología desplegada
Servicio Coolify `rzittzudkunwx8gilonn7tqe` ("oh-daddy", proyecto **AI AGENCY** /
production) — stack compose de 5 contenedores en la red `<uuid>`:
| Contenedor | Imagen | Rol |
|---|---|---|
| `app-<uuid>` | `oh-daddy-app:local` (construida en el server) | Next.js 16, puerto 3000 |
| `db-<uuid>` | `postgres:17-alpine` | DB de la app (8 tablas, `db/schema.sql`) |
| `inngest-<uuid>` | `inngest/inngest:v1.44.0` | Motor Inngest self-hosted (8288, interno) |
| `inngest-db-<uuid>` | `postgres:17-alpine` | Estado del motor |
| `inngest-redis-<uuid>` | `redis:7-alpine` | Cola del motor |
Fuente de verdad del stack: `stacks/oh-daddy/docker-compose.coolify.yml` (sin
secretos; llegan por envs del servicio). Redeploy: `scripts/apps/Deploy-OhDaddy.ps1`.
## Wiring de la app (replica el contract de railway-setup.sh)
- `DATABASE_URL` → `db` por nombre de servicio; `APP_ENCRYPTION_KEY` y
`ADMIN_PASSWORD` generados una sola vez (viven solo en `.env.local.ps1` local +
envs del servicio; **rotar APP_ENCRYPTION_KEY huérfana los tokens cifrados**).
- `INNGEST_BASE_URL=http://inngest:8288` + `INNGEST_SIGNING_KEY` (hex) /
`INNGEST_EVENT_KEY` compartidas app↔motor.
- `NEXT_PUBLIC_APP_URL=https://ohdaddy.urieljareth.org` (build arg + runtime).
- La imagen arranca con `bash scripts/start.sh` (el contract de `railway.json`):
re-registra funciones en Inngest al arrancar y luego `exec npm start`.
- Registro manual: `curl -X PUT https://ohdaddy.urieljareth.org/api/inngest`.
Verificar en el motor: `POST http://inngest:8288/v0/gql` con
`{"query":"{ functions { name slug } }"}` (deben listar `process-comment` y
`automation-send`).
- Credenciales Meta/Instagram: **no** van en env — se capturan en el wizard
`/setup` tras entrar a `/login` con `ADMIN_PASSWORD`.
## Cómo se desplegó (patrón Solo Leveling, imagen local)
1. `New-CoolifyService.ps1 -NoDeploy` creó el servicio (compose base64 + `urls`
→ FQDN del servicio `app`).
2. Un `POST /services/{uuid}/start` (que **falla en el pull** de
`oh-daddy-app:local`, esperado) materializó en disco el compose normalizado
con labels Traefik completos, el `.env` y la red `<uuid>`.
3. Build en el server: clone + `stacks/oh-daddy/Dockerfile` inyectado (multi-stage
node:22-alpine, `NEXT_PUBLIC_APP_URL` como build arg) → `oh-daddy-app:local`.
4. `docker compose up -d` manual + `docker network connect <uuid> coolify-proxy`.
5. `db/schema.sql` aplicado con `docker exec -i db-<uuid> psql` (idempotente).
6. `PUT /api/inngest` público → 200.
## Gotchas nuevos (no documentados antes)
- **`POST /services/{uuid}/envs` da 409** si el compose ya declaró `${VAR}`:
Coolify auto-crea las claves vacías al parsear. Usar **`PATCH
/services/{uuid}/envs/bulk`** con `{"data":[{key,value,is_literal:true}]}`.
- **`GET /deploy?uuid=` da 405 en 4.3.17** — el trigger válido es
`POST /services/{uuid}/start`.
- **El primer re-sync de Inngest del arranque falla con `503 no available
server`**: `start.sh` hace el PUT antes de que Traefik considere healthy el
contenedor. Es benigno — reintentar el PUT cuando la app esté healthy.
- El `wget` de busybox (imagen alpine) soporta `--post-data`/`--header` para
golpear el gql del motor desde el contenedor app.
## Verificación (2026-09-08)
- `GET /resources` → `running:healthy`; 5 contenedores healthy; DNS entre
hermanos OK (`db`, `inngest`, `inngest-db`, `inngest-redis`).
- Motor: `{"data":{"functions":[{"name":"automation-send"...},{"name":"process-comment"...}]}}`,
app "oh-daddy" registrada, `/health` OK.
- `Test-ServiceOnline.ps1 -Fqdn https://ohdaddy.urieljareth.org -Path /login`:
HTTP 200 + render Chromium limpio (título "Oh Daddy. Comment automations on
autopilot"), sin `pageerror`.
- Restart policy `unless-stopped` en los 5 contenedores (sobreviven reinicios
del LXC junto con el autostart de Docker).
## Pendiente humano
Entrar a `https://ohdaddy.urieljareth.org/login` con `ADMIN_PASSWORD` (única
copia en plaintext: el reporte del deploy / `.env.local.ps1`) y completar el
wizard `/setup` con las credenciales de la app de Meta.
@@ -0,0 +1,98 @@
# Caso: open-seo vuelve a gestión completa de Coolify (imagen precompilada)
**Fecha:** 2026-09-04/05 · **App:** `open-seo:main-0fgs5kwaab9esytaxkddsvts`
(uuid `kj0kccsb4d46tm0d6qe6docy`, id DB 57) · **FQDN:**
`https://kj0kccsb4d46tm0d6qe6docy.urieljareth.org`
## Contexto
El sitio público estaba sirviéndose por una **cadena manual improvisada** tras
la caída del 2026-08-27 22:14:
```
Traefik → open-seo-sidecar (nginx:alpine manual, montado 22:25 ese día)
→ proxy_pass → test-openseo (docker run manual de ghcr.io/every-app/open-seo:latest)
```
La aplicación en Coolify quedó `exited:unhealthy` sin contenedor. El usuario
fijó como objetivo: **todo gestionable desde Coolify**.
## Qué se hizo (en orden)
1. **Limpieza de filas huérfanas de env vars** (ids 1152-1156, app 52
`insta-portal`): eran duplicados invisibles de un INSERT SQL manual con el
morph type mal escapado (`App\\Models\\Application`). Las variables reales ya
existían cifradas (ids 1161-1170) → se **borraron** los duplicados, no se
activaron. Backup: `/root/backups/envvar-orphans-1152-1156-20260904-212119.tsv`.
Tras esto: 0 valores sin cifrar en `environment_variables` de toda la instancia.
2. **Deploy git+railpack (intento 1):** `POST /deploy` contra el origen. El
build tardó ~23 min y terminó, pero la app servía **404**: el repo
(`github.com/every-app/open-seo`, público, actualizado ese mismo día) ahora
compila un monorepo (`dist/client|server|open_seo_audit`, sin `index.html`
en raíz) y railpack eligió un plan "estático con Caddy" que no corresponde.
3. **Conversión a imagen precompilada** (doctrina del caso firecrawl):
- API: `docker_registry_image_name=ghcr.io/every-app/open-seo`,
`docker_registry_image_tag=latest` (la API acepta estos campos).
- DB: `UPDATE applications SET build_pack='dockerimage' WHERE id=57` — la
API **rechaza** `build_pack=dockerimage` en PATCH (enum de validación sin
ese valor, 422 "The selected build pack is invalid"), aunque el pipeline
de deploy lo soporta de forma nativa
(`deploy_dockerimage_buildpack` usa `docker_registry_image_name`, no
`static_image`). Backup previo:
`/root/backups/app57-pre-dockerimage-20260904-215456.tsv`.
- Deploy 2: pull de `:latest` + rolling update. El contenedor
**crash-loopeaba (exit 1)**: el preflight de la nueva imagen exige
configurar auth.
4. **Env vars nuevas vía API** (`POST /applications/{uuid}/envs` — cifra por
modelo, sin riesgo del bug de texto plano):
- `AUTH_MODE=local_noauth` — replica el estado previo (el sitio ya corría
público sin auth vía sidecar). Para Cloudflare Access:
`AUTH_MODE=cloudflare_access` + `TEAM_DOMAIN` + `POLICY_AUD`.
- `ALLOWED_HOST=kj0kccsb4d46tm0d6qe6docy.urieljareth.org` — allowlist de
Vite detrás de proxy.
- Deploy 3: contenedor **healthy** (la imagen GHCR trae healthcheck con
`start_period=300s`, a diferencia de los servicios del §1.5 del índice).
5. **Retiro de los contenedores manuales** (con snapshots previos en
`/root/backups/*-inspect-20260904-212317.json`):
- `open-seo-sidecar` (nginx) — además sus labels duplicaban los routers de
Traefik del FQDN y provocaban 503 mientras coexistía con el contenedor nuevo.
- `test-openseo` (backend manual) — ya sin referencias.
## Resultado
```
status=running:healthy
build_pack=dockerimage image=ghcr.io/every-app/open-seo:latest
FQDN → HTTP 200 <title>OpenSEO</title> (servido por el contenedor de Coolify)
```
Dominio, env vars (DATAFORSEO_API_KEY, AUTH_MODE, ALLOWED_HOST), healthcheck,
redeploys y rollbacks: todo administrable desde la UI/API de Coolify.
## Rollback
- **App a git-build:** `UPDATE applications SET build_pack='railpack' WHERE
id=57;` y redeploy (nota: con el main actual vuelve a servir 404 — ver paso 2).
- **Imagen anterior:** el tag local `ghcr.io/every-app/open-seo:sha-c469a48`
(12 días) sigue en el host; o fijar `docker_registry_image_tag` a ese sha.
- **Contenedores manuales:** recrear desde los inspect snapshots (sidecar:
`docker run -d --name open-seo-sidecar --network coolify --restart
unless-stopped -v /tmp/openseo-sidecar/nginx.conf:/etc/nginx/nginx.conf:ro
<labels-del-snapshot> nginx:alpine`).
## Lecciones (añadir a la lista mental de gotchas)
- **PATCH /applications/{uuid} no acepta `build_pack=dockerimage`** aunque el
backend lo soporta y `POST /applications/dockerimage` lo crea así. Para
convertir una app existente: DB o recrear el recurso.
- **`static_image` NO es la imagen del build pack dockerimage** — ese modo lee
`docker_registry_image_name` (+`docker_registry_image_tag`, default `latest`).
- **Un contenedor manual con los labels de Traefik de una app de Coolify
rompe el enrutamiento** cuando la app real vuelve a deployar (routers
duplicados → 503). Al restaurar una app, retirar esos "sidecars con labels".
- La nueva imagen de open-seo exige `AUTH_MODE` en su preflight (exit 1 si
falta) y recomienda `ALLOWED_HOST` detrás de proxy.
@@ -99,7 +99,7 @@ Desde este repo (PowerShell, vía el path SSH habitual):
``` ```
En la configuración activa (línea `INF Updated to new configuration version=N`), verificar que los servicios de 6001 y 6002 digan `http://` y no `https://`. El runbook En la configuración activa (línea `INF Updated to new configuration version=N`), verificar que los servicios de 6001 y 6002 digan `http://` y no `https://`. El runbook
[runbooks/cloudflare-tunnel.md](runbooks/cloudflare-tunnel.md) tiene el procedimiento completo paso a paso. [runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md) tiene el procedimiento completo paso a paso.
--- ---
@@ -1,3 +1,23 @@
> # ⚠️ DOCUMENTO OBSOLETO — NO SEGUIR
>
> Archivado el 2026-08-07. **No uses este archivo como guía.** Contiene datos que
> contradicen la realidad verificada del host:
>
> - Da `192.168.0.117` como IP del servidor Coolify. **El host Proxmox es
> `192.168.0.200`** y todo se alcanza por `ssh [email protected]` +
> `pct exec 102 -- ...`. Ver [../proxmox-inventory.md](../proxmox-inventory.md).
> - Describe editar la configuración del túnel en archivos del host. **El túnel
> es gestionado desde el dashboard de Cloudflare** (el ingress baja del edge);
> editar archivos en el host no tiene efecto.
>
> **Procedimiento vigente:** [../runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md).
> **Herramienta vigente:** `scripts/Invoke-CloudflareApi.ps1` — ver
> [../TOOL-INDEX.md](../TOOL-INDEX.md).
>
> Se conserva solo por el valor histórico del diagnóstico.
---
# Instrucciones para Agente IA — Cloudflare Tunnel + Coolify # Instrucciones para Agente IA — Cloudflare Tunnel + Coolify
**Entorno:** Proxmox VE → LXC CT 102 → Coolify (Docker) → Traefik + cloudflared **Entorno:** Proxmox VE → LXC CT 102 → Coolify (Docker) → Traefik + cloudflared
+20
View File
@@ -0,0 +1,20 @@
# Incidentes — archivo histórico
> ⚠️ **Esto NO es fuente de verdad.** Es material histórico: incidentes ya
> resueltos y documentos superados. Describe cómo estaba el sistema en la fecha
> de cada archivo, no cómo está hoy.
>
> - Estado actual → [../proxmox-inventory.md](../proxmox-inventory.md)
> - Procedimientos vigentes → [../runbooks/](../runbooks/)
> - Qué herramienta usar → [../TOOL-INDEX.md](../TOOL-INDEX.md)
Se conserva porque el diagnóstico y la causa raíz siguen siendo útiles cuando un
síntoma parecido reaparece.
| Archivo | Fecha | Qué fue | Estado |
|---|---|---|---|
| [2026-04-11-cloudflare-tunnel-websocket-tls.md](2026-04-11-cloudflare-tunnel-websocket-tls.md) | 2026-04-11 | Terminal de Coolify y websockets de la UI caídos: routing del túnel en los puertos 6001/6002. | Resuelto. Procedimiento vigente en [../runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md). |
| [2026-04-11-coolify-static-app-deploy.md](2026-04-11-coolify-static-app-deploy.md) | 2026-04-11 | App estática desde GitHub servía página en blanco y 502. | Resuelto. Era Coolify v4.0.0-beta.472; hoy corre v4.1.2. |
| [2026-06-29-coolify-cleanup-report.md](2026-06-29-coolify-cleanup-report.md) | 2026-06-29 | Limpieza de imágenes, volúmenes y redes de Docker (−4.7 GB). | Reporte de una ejecución puntual. |
| [2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md](2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md) | 2026-04 | Guía de agente para el túnel Cloudflare. | **Obsoleta y contradictoria** — ver el aviso dentro del archivo. Usa el runbook. |
| [openclaw.md](openclaw.md) | varias | Historial de incidentes de OpenClaw. | Referencia. |
+177 -28
View File
@@ -1,46 +1,195 @@
# Proxmox Inventory # Inventario Proxmox — fuente de verdad del estado
Last verified: 2026-05-30 local time. **Última verificación: 2026-08-29** (SSH, API de Proxmox y API de Coolify desde esta máquina).
Este documento es la fuente de verdad de *qué existe*. Si algo aquí contradice a
la realidad del host, gana el host: re-verifica y actualiza este archivo.
Para refrescarlo:
```powershell
. .\.env.local.ps1
.\scripts\Get-ProxmoxInventory.ps1
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
Sort-Object name | Select-Object name, uuid, fqdn
```
---
## Host ## Host
- IP: `192.168.0.200` | Dato | Valor |
- Node: `thinkcentre` |---|---|
- Proxmox VE: `9.1.1` | IP | `192.168.0.200` (IPs verificadas en red: `192.168.3.23` / `192.168.3.15` / `192.168.0.200`) |
- Running kernel: `6.17.2-1-pve` | Nodo | `thinkcentre` |
- Topology: single-node Proxmox host | Proxmox VE | `9.1.1` (`pve-manager/9.1.1/42db4a6cf33dac83`) |
| Kernel | `6.17.2-1-pve` |
| Topología | Proxmox de un solo nodo |
| Disco `/` | 39 GB, 11 GB usados (**29%**) |
| RAM | 31 GiB totales — 14 GiB en uso, 16 GiB disponibles |
| Swap | 7.6 GiB, 303 MiB en uso |
## LXC containers ## Contenedores LXC
| VMID | Name | Status | Notes | | VMID | Nombre | Estado | IPs | Notas |
| --- | --- | --- | --- | |---|---|---|---|---|
| 100 | hermes | running | Secondary LXC | | 100 | hermes | **running** | `192.168.3.23` / `192.168.3.15` | Secundario. Actualizado a commit `5d3c15aaa`. MiniMax-M3 y API key configuradas y verificadas con inferencia CLI en vivo. Ver caso [docs/casos/hermes-minimax-m3-setup.md](casos/hermes-minimax-m3-setup.md). |
| 102 | coolify | running | Docker host for Coolify and apps | | 102 | coolify | running | `192.168.0.200` (host bridge) | Host Docker de Coolify y de todas las apps. |
No QEMU VM was listed during the latest smoke test. **No hay VMs QEMU** (`qm list` vacío).
## Docker inside LXC 102 ## Docker dentro de LXC 102
Docker is not managed directly on the Proxmox host. Use: Docker **no** se administra en el host Proxmox directamente. Todo comando va
envuelto:
```powershell ```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps"
``` ```
Observed groups: **68 contenedores, todos `running`** al momento de la verificación.
- Coolify core: `coolify`, `coolify-db`, `coolify-redis`, > **Los nombres de contenedor llevan el uuid de Coolify como sufijo** — no son
`coolify-realtime`, `coolify-sentinel`, `coolify-proxy` > adivinables y **cambian si un redeploy recrea el contenedor**. Resuélvelos
- Tunnel/proxy: `cloudflared` > siempre antes de usarlos; ver §1.3 de [TOOL-INDEX.md](TOOL-INDEX.md).
- Apps currently observed: Gitea, Chatwoot, OpenClaw, browser services, n8n,
CodiMD, Qdrant, Baserow, Grimmory
One Baserow container was observed as `health: starting` during the latest ### Núcleo de la plataforma
Docker sample, so recheck before assuming it is unhealthy.
## Access model | Contenedor | Imagen |
|---|---|
| `coolify` | `ghcr.io/coollabsio/coolify:4.3.14` (`GET /version` → `4.3.14`, verificado 2026-08-29) |
| `coolify-db` | `postgres:15-alpine` |
| `coolify-redis` | `redis:7-alpine` |
| `coolify-realtime` | `ghcr.io/coollabsio/coolify-realtime:1.0.17` |
| `coolify-sentinel` | `ghcr.io/coollabsio/sentinel:0.0.22` |
| `coolify-proxy` | `traefik:v3.6` |
| `cloudflared` | — (túnel gestionado desde el dashboard) |
- SSH: root over key-based auth. ### Recursos registrados en Coolify (29)
- API: Proxmox token via `PROXMOX_API_TOKEN_ID` and
`PROXMOX_API_TOKEN_SECRET`. Del endpoint `/resources`. El **uuid** es lo que necesitas para la API y para
- Secrets must stay in local env files or the OS secret store, not Markdown. resolver nombres de contenedor.
| Recurso | uuid | FQDN |
|---|---|---|
| agendamax:main | `s30f7egdlkx4wyjp59o1iunc` | https://agendamax.urieljareth.org |
| audio-a--texto:main | `up0oaqnpd8kywjkmmihb61t0` | https://stt.urieljareth.org |
| baserow:main | `vngcvnhbfqboov4nln88zc73` | — |
| baserow-redis | `vztgldo9cap3s0oj240tokgj` | — |
| chatwoot | `c11xzy2tx2cdapm32f5b89vy` | — |
| codimd | `ag4ndg4cr1hvczkr35qlyzs8` | — |
| cotizador:main | `e6vie34d83iv5vw3eyzinl3s` | `…:3000` (sin dominio propio) |
| demospa | `2f094556a04c0dee043af215` | https://demospa.urieljareth.org |
| demospa-mysql | `teplxg70a97itolgwk2qkmgi` | — |
| e3-manager-demo | `xxoq93pnz0jta5tz5m50ylig` | — |
| estaci-n-de-documentos:main | `u11ug3eizk9du3p2ch556ud4` | — |
| **evolution-go** | `j0jkacsfcgypm2jmpillls01` | https://evo.urieljareth.org (API WhatsApp 0.7.2, licencia activa — ver [casos/evolution-go-stack.md](casos/evolution-go-stack.md)) |
| evolution-api | `q6tnsvkvrjw4g0ab532l3r1s` | https://evoapi.urieljareth.org (Node v2.3.7; integrada a Chatwoot: inbox 6=Asesoría, 10=Personal, 12=JM; INSTA desconectada — ver [casos/evolution-api-chatwoot.md](casos/evolution-api-chatwoot.md)) |
| **firecrawl:main** | `du3iknyvy22vap767t9tnf9s` | — (**registrado pero sin contenedor corriendo**) |
| gitea-with-postgresql | `hjwh0svsoo9p5w5kj2j6b1bd` | — |
| grimmory | `y8cq6jmboz0b22mn61hs4tu8` | — |
| insta-portal | `instademo0portal0insta0demo1` | `…:4180` |
| n8n-with-postgres-and-worker | `jdj3y3kmz9blec7ntbxuhezi` | — |
| nextcloud-with-postgres | `hdcdpkm0jko3qqvn5683ercc` | — |
| openclaw-business | `zhaz04q8ibqp5r5hz5ibo01t` | — |
| openclaw (2ª instancia) | `uudgcoz5ibvyvullyzbjalai` | — |
| open-webui | `q13zdxusnhvdent7f44a18kc` | — |
| plataforma-interna | `kruadlc7fdrbh28ykrv8rdyl` | — |
| postgresql-database | `j6hgnvdwq9anpa9wj2ikdx0j` | — |
| postgresql-database | `s10bvby71tb1fpbcl4tnxzzh` | — |
| postgresql-database | `tflojv1iqh0ueoq7apxx4mos` | — |
| prompt-gallery-e3:main | `39eu76lqbejdy9vxs5iz1rws` | https://demopromptgallerye3.urieljareth.org |
| qdrant | `uyn0js6pqbwo8mubw5edy95f` | — |
| solo-leveling | `urm8m4u0jvjggmgpfxblnqwc` | — |
### Stack de Supabase
Corre con nombres fijos (sin sufijo uuid), fuera del patrón habitual de Coolify:
`supabase-db` (`supabase/postgres:17.6.1.136`), `supabase-kong`,
`supabase-auth`, `supabase-rest`, `supabase-storage`, `supabase-studio`,
`supabase-meta`, `supabase-imgproxy`, `realtime-dev.supabase-realtime`.
También `e3-mailpit` (`axllent/mailpit`).
### Versiones que importan
| App | Imagen en ejecución |
|---|---|
| **Chatwoot** | **`chatwoot/chatwoot:v4.16.2`** (app y sidekiq) |
| Chatwoot DB | `pgvector/pgvector:pg12` |
| n8n | `n8nio/n8n:2.32.7` |
| Baserow | `baserow/baserow:2.3.2` |
| OpenClaw | `coollabsio/openclaw:2026.7.1` |
| Evolution API | `evoapicloud/evolution-api:v2.3.7` |
| Gitea | `gitea/gitea:latest` |
| Nextcloud | `lscr.io/linuxserver/nextcloud:latest` |
| Grimmory | `grimmory/grimmory:nightly` |
| Solo Leveling | `ghcr.io/urieljarethbusiness-cpu/solo-leveling:latest` |
## Estado de la API de Coolify
Base pública: `https://coolify.urieljareth.org/api/v1` (Bearer `COOLIFY_TOKEN`) ·
**Origen: `http://192.168.0.117:8000/api/v1`** · Instancia: **4.3.14**.
**Re-verificado a fondo el 2026-08-29, con hallazgo que corrige todo lo anterior:**
- **La API de esta instancia está COMPLETA.** Su propio `openapi.yaml` (dentro del
contenedor en `/var/www/html/openapi.yaml`) declara el namespace completo de
`/applications/*` (incl. `POST /applications/public`, `/dockerfile`,
`/private-deploy-key`, `/private-github-app`), `/github-apps`, `/gitlab-apps`,
notificaciones, proveedores cloud (Hetzner/DigitalOcean/Vultr), MCP, etc. — y
contra el **origen** esos endpoints responden 200 con el token de siempre.
- **El 404 de `/applications/*` que se venía documentando desde v4.1.2 NO lo
produce Coolify: lo produce Cloudflare en el hostname público.** Mismo token,
misma ruta: `https://coolify.urieljareth.org/api/v1/applications` → 404;
`http://192.168.0.117:8000/api/v1/applications` → 200. Es un bloqueo del edge
(regla WAF/ruta en el dashboard) que hay que corregir allí; mientras tanto,
llama a esos endpoints contra `COOLIFY_API_URL_ORIGIN`.
- **Desde v4.2 los endpoints de estado exigen POST** (`GET /deploy?uuid=` → 405
`"This endpoint has changed to a POST request."` — confirmado también contra el
origen; ver §10.1 de las notas). En 4.3.x se añadieron endpoints de logs
(db/servicio/contenedor), settings en las respuestas de application y MCP.
| Endpoint (con token válido) | Vía Cloudflare | Vía origen (LXC 102) |
|---|---|---|
| `/version`, `/resources`, `/servers`, `/projects`, `/teams`, `/services`, `/databases`, `/deployments`, `/security/keys` | OK | OK |
| **`/applications` y todo su namespace** | **404 (Cloudflare)** | **OK (200)** |
| **`/github-apps`** | **404 (Cloudflare)** | **OK (200)** |
| `/deploy` con GET | 405 | 405 (correcto: exige POST desde v4.2) |
## Modelo de acceso
- **SSH:** root con autenticación por clave. La llave operativa es
`keys/proxmox_ed25519` (idéntica a `C:\Users\Uriel Jareth\.ssh\coolify_key`;
ED25519, sin passphrase) — verificada en vivo el 2026-08-29 contra el host
(`192.168.0.200`) y el LXC 102 (`192.168.0.117`). La ruta
`…\.openclaw\workspace\proxmox_key_win` citada en docs antiguos **ya no existe**.
Es el único camino directo al host; permite ejecutar
comandos en el host y en los contenedores LXC (`pct exec 100`, `pct exec 102`).
- **API de Proxmox:** token `root@pam!openclaw` — único token del host
(`/etc/pve/priv/token.cfg`), verificado 200 el 2026-08-29. Ya cargado en
`.env.local.ps1` (`PROXMOX_API_TOKEN_ID` / `PROXMOX_API_TOKEN_SECRET`).
- **API de Coolify:** `COOLIFY_TOKEN` (v4.3.14, verificado).
- **API de Cloudflare:** sin token — el túnel se gestiona desde el dashboard.
- Los secretos viven solo en `.env.local.ps1` (gitignored), en `ACCESS.md`
(gitignored, fuente de verdad de credenciales) o en el almacén de secretos del
SO. **Nunca en Markdown versionado.** Inventario de credenciales:
[ACCESS.md](../ACCESS.md)
> ⚠️ **Pendiente en `.env.local.ps1` (2026-08-29):** solo `CLOUDFLARE_API_TOKEN`.
> El PAT de GitHub que había caducó (401); se reemplazó por la credencial viva del
> Administrador de credenciales de Windows. `PROXMOX_API_TOKEN_ID/SECRET`,
> `COOLIFY_EMAIL/PASSWORD` ya están cargados. SSH, API de Proxmox y API de
> Coolify verificados.
## Automatizaciones instaladas en el host
| Qué | Dónde | Agendado por |
|---|---|---|
| Guard del parche enterprise de Chatwoot | `/root/scripts/chatwoot-enterprise-guard.sh` | `/etc/cron.d/chatwoot-enterprise-guard`, `*/5 * * * *` |
| Auto-arranque tras corte de luz | unit `coolify-autostart.service` | systemd, **`enabled`** |
Log del guard: `/var/log/chatwoot-enterprise-guard.log` (solo escribe cuando
actúa). Estado al 2026-08-07: el plan está en `enterprise` y una ejecución
manual del guard pasa correctamente, pero el log registra errores de lectura
durante la ventana del update a v4.16.2 — ver
[runbooks/chatwoot-update.md](runbooks/chatwoot-update.md).
+111 -10
View File
@@ -31,17 +31,41 @@ de Cloudflare arranquen solos**, sin intervención manual.
En cada boot el guardián: En cada boot el guardián:
1. Verifica que el LXC 102 esté `running` (lo arranca si no). 1. Verifica que el LXC 102 esté `running` (lo arranca si no).
2. Espera a que Docker responda dentro del LXC (hasta 180 s). 2. Espera a que Docker responda dentro del LXC (**deadline de reloj real**,
3. Se asegura de que estén arriba: `coolify-db`, `coolify-redis`, `COOLIFY_MAX_WAIT`, por defecto 600 s).
`coolify-realtime`, `coolify`, `coolify-proxy`, `cloudflared` (los inicia si 3. Da un margen de asentamiento (`COOLIFY_SETTLE`, 180 s) para que Docker
alguno no está). arranque sus propios contenedores, y solo entonces fuerza el arranque de
los que falten: `coolify-db`, `coolify-redis`, `coolify-realtime`,
`coolify`, `coolify-proxy`, `cloudflared`.
4. Levanta el `cloudflared.service` de systemd dentro del LXC (segundo 4. Levanta el `cloudflared.service` de systemd dentro del LXC (segundo
conector al mismo túnel). conector al mismo túnel).
5. **Comprueba que el túnel realmente llegó a Cloudflare**: cuenta los
`Registered tunnel connection` de este boot y los escribe en el log.
Sale con código ≠ 0 si algo no se pudo arrancar, para que systemd lo marque
`failed` y `Restart=on-failure` reintente (hasta 3 veces por hora).
Las dos capas son complementarias: `onboot` hace el trabajo normal; el guardián Las dos capas son complementarias: `onboot` hace el trabajo normal; el guardián
es una red de seguridad que además **auto-repara** (p. ej. un contenedor con es una red de seguridad que además **auto-repara** (p. ej. un contenedor con
`restart=no`) y deja **log** de lo ocurrido tras el apagón. `restart=no`) y deja **log** de lo ocurrido tras el apagón.
### Presupuestos de tiempo (no los bajes a ciegas)
Medido en el boot del 2026-08-07 16:41: `pve-guests` tarda **78 s** en arrancar
el CT 102, el daemon de Docker dentro del LXC solo responde ~**4-5 min** después
del encendido, y el último contenedor core (`coolify`) arranca a los **7 m 40 s**.
Por eso:
| Parámetro | Valor | Regla |
|---|---|---|
| `COOLIFY_MAX_WAIT` | 600 s | espera de Docker, por **reloj real** |
| `COOLIFY_SETTLE` | 180 s | margen antes de forzar arranques |
| `COOLIFY_PROBE_TIMEOUT` | 20 s | timeout duro de **cada** llamada al LXC |
| `TimeoutStartSec` (unit) | 1200 s | **debe superar** `MAX_WAIT + SETTLE` |
`COOLIFY_LOG` también es sobreescribible, para poder hacer pruebas en seco sin
tocar el log de producción.
Los archivos fuente viven en el repo en [scripts/host/](../../scripts/host/) y se Los archivos fuente viven en el repo en [scripts/host/](../../scripts/host/) y se
instalan con [scripts/Install-CoolifyAutostart.ps1](../../scripts/Install-CoolifyAutostart.ps1). instalan con [scripts/Install-CoolifyAutostart.ps1](../../scripts/Install-CoolifyAutostart.ps1).
@@ -78,10 +102,17 @@ Estado sano esperado:
``` ```
onboot: 1 onboot: 1
startup: order=1,up=30 startup: order=1,up=30
guardian enabled: enabled enabled: enabled
script executable: yes state: active
result: success
timeout: 20min
``` ```
**No basta con `enabled`.** `enabled` solo dice que arrancará; `state`/`result`
dicen si la última ejecución funcionó. Una corrida sana del log termina en
`=== coolify-autostart done (failures=0) ===`. Si el log se corta justo después
de `LXC 102 already running`, el guardián murió esperando a Docker.
Log del último arranque: Log del último arranque:
```powershell ```powershell
@@ -107,6 +138,63 @@ Para validar que el unit está bien formado y en el orden correcto:
`After` debe incluir `pve-guests.service`; `WantedBy` debe ser `multi-user.target`. `After` debe incluir `pve-guests.service`; `WantedBy` debe ser `multi-user.target`.
Para probar la ruta de **fallo** (que el guardián corte y deje `ERROR` en vez de
colgarse), apúntalo a un LXC inexistente con un log temporal:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "env COOLIFY_LXC=999 COOLIFY_LOG=/tmp/ca-test.log COOLIFY_MAX_WAIT=15 COOLIFY_PROBE_TIMEOUT=5 bash /usr/local/bin/coolify-autostart.sh > /dev/null 2>&1; cat /tmp/ca-test.log; rm -f /tmp/ca-test.log"
```
Debe terminar en `ERROR: docker not ready after 15s (wall clock) -> aborting`
en ~18 s. No toca producción ni el log real.
---
## Incidente 2026-08-07 — el guardián llevaba 2/2 arranques muriendo
**Síntoma:** `systemctl is-enabled` decía `enabled`, pero la unidad estaba
`failed (Result: timeout)` en los dos reinicios reales del día (09:47 y 16:41).
El log se cortaba siempre en `LXC 102 already running`, sin línea de `ERROR`.
**Cronología del boot de las 16:41:**
| Hora | Evento |
|---|---|
| 16:41:49 | `pve-guests` arranca el CT 102 |
| 16:43:07 | termina `pve-guests` (78 s) y arranca el guardián |
| 16:43:08 | `LXC 102 already running` → entra a esperar Docker |
| 16:46:05 | Docker empieza a levantar contenedores (`coolify-db`) |
| **16:48:07** | **systemd mata al guardián**: `TimeoutStartSec=300` |
| 16:48:40 | arranca `coolify` — **33 s después de que el guardián ya estaba muerto** |
**Causa raíz (tres defectos que se sumaron):**
1. El bucle de espera contaba **iteraciones de `sleep`, no reloj real**, así que
`MAX_WAIT=180` no acotaba nada.
2. Las llamadas `pct exec ... docker info` **no tenían timeout** y `docker info`
es caro (enumera los ~50 contenedores): con el daemon saturado en el arranque
en frío, una sola llamada se bloqueaba minutos y consumía todo el presupuesto
en silencio.
3. `TimeoutStartSec=300` estaba **por debajo del tiempo real de convergencia**
(~7 m 40 s), así que systemd mataba al guardián antes de que pudiera actuar.
Nunca hubo una ejecución exitosa en un arranque real: la única corrida sana del
log (2026-07-08) fue el `-RunNow` manual con todo ya arriba.
**Por qué no se notó durante un mes:** `-VerifyOnly` solo miraba `is-enabled`.
Ahora también reporta `state`, `result`, `timeout` y el journal del último boot.
**Qué salvó el servicio mientras tanto:** las capas base, que sí funcionaron en
los dos reinicios — `onboot=1`, `docker.service` `enabled`, políticas
`restart=always`/`unless-stopped` y `cloudflared.service` `enabled` (registró sus
4 conectores QUIC a los 3 m 41 s del boot). El guardián es red de seguridad, no
el mecanismo principal; por eso el apagón no se notó de cara al usuario.
**Corrección:** deadline por reloj real, `timeout` duro en cada llamada al LXC,
sonda barata `docker version` en vez de `docker info`, margen de asentamiento
antes de forzar arranques, verificación de conectores del túnel en el log,
`TimeoutStartSec=1200` y `Restart=on-failure`.
--- ---
## Rollback ## Rollback
@@ -133,10 +221,23 @@ Para validar que el unit está bien formado y en el orden correcto:
[cloudflare-tunnel.md](cloudflare-tunnel.md). [cloudflare-tunnel.md](cloudflare-tunnel.md).
- `coolify-sentinel` tiene `restart=no` (monitor no crítico); Coolify lo recrea, - `coolify-sentinel` tiene `restart=no` (monitor no crítico); Coolify lo recrea,
por eso no está en la lista de contenedores core del guardián. por eso no está en la lista de contenedores core del guardián.
- **El único LXC con `onboot` es el 102.** El LXC `100 hermes` no tiene la marca,
así que **no** arranca solo tras un apagón. Es intencional mientras sea
secundario; si algún día deja de serlo, `pct set 100 --onboot 1`.
--- ---
**Verificado:** 2026-07-08 — instalado y probado en vivo. `onboot=1`, **Verificado:** 2026-08-07 — auditoría completa tras dos reinicios reales del
`coolify-autostart.service` `enabled`, guardián ejecutado con éxito (LXC arriba, día. Se detectó y corrigió el fallo del guardián (ver incidente arriba). Estado
Docker listo, 6 contenedores core + túnel `running`). `systemd-analyze verify` sin final: `onboot=1`, unidad `enabled` / `active` / `result=success`,
warnings. `TimeoutStartUSec=20min`, `Restart=on-failure`, `After` incluye
`pve-guests.service`, `systemd-analyze verify` sin warnings. Guardián ejecutado
de punta a punta en 9 s con `failures=0`, 6 contenedores core `running`,
`cloudflared.service` `active` con **4 conectores registrados**. Ruta de aborto
por reloj real probada en seco (corta a los 18 s con `ERROR`). Público
verificado a través del túnel: `coolify.urieljareth.org` → 302,
`chat.urieljareth.org` → 200.
**Verificado:** 2026-07-08 — instalación original (`onboot=1` + guardián). La
prueba de entonces fue un `-RunNow` manual, no un arranque real; de ahí que el
defecto de tiempos no se detectara hasta 2026-08-07.
+580
View File
@@ -0,0 +1,580 @@
# Runbook: actualizar Chatwoot en Coolify sin perder la edicion enterprise
> **Ejecutado end-to-end el 2026-07-24.** Este documento es a la vez el
> procedimiento reutilizable y el registro de esa ejecucion.
> Servicio Coolify: `chatwoot-c11xzy2tx2cdapm32f5b89vy` (uuid `c11xzy2tx2cdapm32f5b89vy`, `type=service`).
> FQDN real: **`https://chat.urieljareth.org`** (el FQDN que aparece en
> [docs/casos/chatwoot-enterprise-patch.md](../casos/chatwoot-enterprise-patch.md)
> quedo obsoleto).
> Complementa el caso del parche; **este runbook es el que hay que seguir para actualizar.**
## 0. Resultado de la ejecucion del 2026-07-24
| Dato | Antes | Despues |
|---|---|---|
| Version | 4.16.0 | **4.16.1** |
| Tag de imagen | `chatwoot/chatwoot:latest` (sin pin) | **`chatwoot/chatwoot:v4.16.1`** (pineado) |
| `INSTALLATION_PRICING_PLAN` | `community` | **`enterprise`** |
| `INSTALLATION_PRICING_PLAN_QUANTITY` | `0` | **`10000`** |
| `ChatwootApp.self_hosted_enterprise?` | `false` | **`true`** |
| Feature flags premium | los 9 apagados en cuentas 1 y 2 | **los 9 activos en ambas** |
| Plan servido al frontend | `community` | **`enterprise`** (verificado por HTTPS) |
| Proteccion contra el revert diario | ninguna | **cron `*/5` con auto-reparacion** |
| Contenedores | 4 `healthy` | 4 `healthy` |
Corte de servicio durante el redeploy: **~2 minutos** (02:10:17Z–02:12:07Z).
Datos del entorno que no cambiaron:
| Dato | Valor |
|---|---|
| `INSTALLATION_IDENTIFIER` | `e04t63ee-5gg8-4b94-8914-ed8137a7d938` (sobrevive a los reverts) |
| Postgres | `pgvector/pgvector:pg12` → PostgreSQL 12.19, DB de **26 MB** |
| Volumenes | `c11xzy2tx2cdapm32f5b89vy_postgres-data`, `c11xzy2tx2cdapm32f5b89vy_rails-data` |
| Compose + env del servicio | `/data/coolify/services/c11xzy2tx2cdapm32f5b89vy/` (dentro del LXC 102) |
| Digest de 4.16.0 (para rollback) | `sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175` |
| Digest de 4.16.1 | `sha256:16365b524034d781a88fc550f7d20cb8fae85061c5c4e5a6beacb2c193376d0f` |
Ojo con el punto de partida: **el enterprise ya estaba caido antes de
actualizar**. La actualizacion no lo tiro; ya estaba tirado (ver seccion 1).
### Pre-flight que hizo la actualizacion de bajo riesgo
`v4.16.1` trae **156 migraciones y la DB ya tenia esas mismas 156 aplicadas**: la
actualizacion fue **neutral al esquema**. Por eso el rollback a 4.16.0 no
necesitaria restaurar la DB. Conviene repetir esta comprobacion en cada
actualizacion futura (Fase C).
### Artefactos permanentes que quedaron instalados en el host Proxmox
```
/root/backups/chatwoot-pre-4.16.1.dump 667K, 97 tablas, sha256 dc3d25fe...
/root/backups/chatwoot-service-pre-4.16.1.tgz 2.4K, compose + .env (tiene secretos)
/root/scripts/chatwoot-enterprise-guard.sh el guard (copia versionada en scripts/)
/etc/cron.d/chatwoot-enterprise-guard cron */5
/etc/logrotate.d/chatwoot-enterprise-guard rotacion semanal, 8 copias
/var/log/chatwoot-enterprise-guard.log solo escribe cuando repara
```
## 1. Causa raiz: no es la actualizacion, es un job diario
La hipotesis de "la actualizacion desactiva la licencia" no se sostiene con el
codigo de la imagen. La cadena real, leida de la imagen 4.16.0:
```
config/schedule.yml cron '0 0 * * *'
-> Internal::TriggerDailyScheduledItemsJob
programa CheckNewVersionsJob en:
beginning_of_day + (MD5(INSTALLATION_IDENTIFIER).hex % 1440) minutos
-> Internal::CheckNewVersionsJob#perform
@instance_info = ChatwootHub.sync_with_hub # POST https://hub.2.chatwoot.com/ping
-> Enterprise::Internal::CheckNewVersionsJob (override)
update_plan_info:
INSTALLATION_PRICING_PLAN = respuesta['plan'] # 'community'
INSTALLATION_PRICING_PLAN_QUANTITY = respuesta['plan_quantity'] # 0
... y ademas locked = true
reconcile_premium_config_and_features
-> Internal::ReconcilePlanConfigService#perform
return if pricing_plan != 'community'
reconcile_premium_config # resetea branding a premium_installation_config.yml
reconcile_premium_features # account.disable_features!(*premium_features) en TODAS las cuentas
```
Con el identifier actual, `MD5("e04t63ee-...").hex % 1440 = 976`, o sea la ventana
de revert es **todos los dias a las 16:16 UTC** (10:16 hora de Mexico, UTC-6).
Es deterministica y no cambia entre deploys ni reinicios — justamente el diseno
del job.
Consecuencias practicas:
1. **El parche caduca en <= 24 h**, actualices o no. El caso se documento el
2026-06-16; se revirtio al dia siguiente.
2. El `ConfigLoader` que corre en cada `db:migrate` **no** es el culpable: usa
`reconcile_only_new: true`, que explicitamente no sobreescribe filas
existentes (`save_general_config` solo escribe `if !@reconcile_only_new`).
3. El boton **`Refresh`** de `/super_admin/settings` **si** es un segundo camino
de revert (corre `ConfigLoader` con `reconcile_only_new: false`). La
advertencia del caso original sigue vigente.
4. Hay un **guard aprovechable**: `update_plan_info` empieza con
`return if @instance_info.blank?`. Si el hub no responde, no se escribe nada.
Esa es la base del fix durable de la Fase G.
## 2. Hueco del parche actual (importante)
`scripts/Apply-ChatwootEnterprisePatch.ps1` corregia **solo 3 filas** de
`installation_configs`. Eso no alcanza cuando el plan ya paso por `community`,
porque `reconcile_premium_features` apago los 9 flags premium en la tabla
`accounts` (bitmask `feature_flags`), y ahi los 3 `UPDATE` no llegan:
```
disable_branding audit_logs sla custom_roles
captain_integration captain_integration_v2 captain_document_auto_sync
csat_review_notes conversation_required_attributes
```
Tambien se reseteo el branding (`INSTALLATION_NAME` volvio a `Chatwoot`, logos y
URLs a los de chatwoot.com, `DISPLAY_MANIFEST` a `true`).
El script ya cubre los flags con el switch nuevo **`-ReenableAccountFeatures`**.
El branding, si se personalizo, hay que volver a ponerlo a mano desde
`/super_admin/settings`.
### 2.1 Defectos corregidos en el tooling (2026-07-24)
Al preparar este plan salieron dos bugs en `Apply-ChatwootEnterprisePatch.ps1`
que habrian hecho fallar los pasos de la Fase E:
1. **La autodeteccion del contenedor Postgres nunca funciono.** El patron era
`"<uuid>.*(pgvector|postgres|db)"`, pero Coolify nombra los contenedores
`<servicio>-<uuid>` (`postgres-c11xzy...`), o sea el uuid va al final. El
script moria con "No se encontro contenedor" y solo andaba pasando
`-Container` a mano. Ahora son dos greps encadenados (uuid, luego rol).
2. **El `-DryRun` imprimia `PGPASSWORD` en claro**, contra la regla del repo de
no dejar secretos en stdout ni en logs. Ahora sale enmascarada.
3. **El parche no invalidaba el cache de `GlobalConfig`.** Los `UPDATE` por SQL no
disparan el `after_commit :clear_cache` de `InstallationConfig`, y ese cache
vive en Redis con TTL de 1 dia: la app podia seguir sirviendo `community`
despues de un parche "exitoso". Ahora el paso de `rails runner` llama
explicitamente a `GlobalConfig.clear_cache` (detalle en la Fase G).
Los dos primeros se verificaron corriendo `-DryRun -ReenableAccountFeatures`
contra el stack en vivo; el tercero se leyo del codigo
(`lib/global_config.rb` + `app/models/installation_config.rb`) y se confirmo
inspeccionando las claves `V1:GLOBAL_CONFIG:*` en Redis.
`Get-ChatwootLicenseStatus.ps1` quedo probado end-to-end.
## 3. Plan de actualizacion
Todo desde la raiz del repo, en PowerShell, con `. .\.env.local.ps1` cargado.
### Fase A — pre-checks (solo lectura)
```powershell
. .\.env.local.ps1
# Estado de licencia + flags por cuenta + ventana diaria de revert.
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
# Salud del stack.
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All
```
Anota el digest de la imagen en uso; es la unica ruta de rollback rapido:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker images --digests | grep chatwoot/chatwoot"
```
> Digest al 2026-07-24: `sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175` (= 4.16.0).
### Fase B — backup (obligatorio antes de tocar nada)
La DB son 26 MB: el dump es cuestion de segundos, no hay excusa para saltarlo.
```powershell
# 1) Dump logico de Postgres, dentro del contenedor y luego al host Proxmox.
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD=\$POSTGRES_PASSWORD pg_dump -U \$POSTGRES_USER -d \$POSTGRES_DB -Fc -f /tmp/chatwoot-pre-4.16.1.dump'"
# 2) Sacarlo del contenedor al LXC y del LXC al host.
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp postgres-c11xzy2tx2cdapm32f5b89vy:/tmp/chatwoot-pre-4.16.1.dump /root/chatwoot-pre-4.16.1.dump"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct pull 102 /root/chatwoot-pre-4.16.1.dump /root/backups/chatwoot-pre-4.16.1.dump && ls -lh /root/backups/"
# 3) Copia del compose + .env del servicio (queda EN EL HOST, nunca en el repo:
# el .env tiene SECRET_KEY_BASE y las passwords de Postgres/Redis).
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- tar czf /root/chatwoot-service-pre-4.16.1.tgz -C /data/coolify/services/c11xzy2tx2cdapm32f5b89vy ."
```
Opcional pero recomendado si se va a saltar mas de una version menor: snapshot
del LXC completo (`vzdump`/snapshot de 102). Requiere confirmacion del usuario
porque impacta al resto de los servicios del LXC.
### Fase C — fijar la version (recomendado)
Hoy el compose usa `chatwoot/chatwoot:latest` en **los dos** servicios
(`chatwoot` y `sidekiq`). Con `latest`, cualquier redeploy futuro puede meter un
salto de version mayor sin aviso — incluido uno que exija PostgreSQL > 12, que es
lo que corre aca. Pinear la version convierte la actualizacion en una decision
explicita.
Primero confirmar que el tag existe. **El naming es `vX.Y.Z`**: `v4.16.1` existe,
`4.16.1` (sin la `v`) no.
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker manifest inspect chatwoot/chatwoot:v4.16.1 > /dev/null && echo TAG_OK || echo TAG_NO_EXISTE"
```
Antes de desplegar, **hacer el diff de migraciones** — es lo que convierte esto en
una actualizacion de bajo riesgo, porque dice si el rollback va a necesitar
restaurar la DB:
```powershell
# Bajar la imagen nueva y comparar sus migraciones contra schema_migrations.
# 156 == 156 significa que no hay cambios de esquema (fue el caso de 4.16.1).
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker pull chatwoot/chatwoot:v4.16.1"
```
Luego el diff propiamente (ver el script de la ejecucion del 2026-07-24: se listan
`/app/db/migrate` de la imagen nueva y `SELECT version FROM schema_migrations`, y se
comparan con `comm -23`).
Para pinear el tag, **usar la API, no editar el archivo en el LXC**: Coolify
regenera `/data/coolify/services/<uuid>/docker-compose.yml` en cada deploy y un
cambio a mano en el host se pierde.
```powershell
. .\.env.local.ps1
# 1) Traer el compose actual, 2) cambiar las 2 lineas de imagen, 3) PATCH en base64.
$svc = (.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services/c11xzy2tx2cdapm32f5b89vy") | ConvertFrom-Json
$new = $svc.docker_compose_raw.Replace("image: 'chatwoot/chatwoot:latest'", "image: 'chatwoot/chatwoot:v4.16.1'")
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($new))
$body = @{ docker_compose_raw = $b64 } | ConvertTo-Json -Compress
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method PATCH -Path "/services/c11xzy2tx2cdapm32f5b89vy" -BodyJson $body
```
> **`docker_compose_raw` tiene que ir en base64.** Mandarlo en texto plano
> devuelve `422 Unprocessable Entity` con
> `"The docker_compose_raw should be base64 encoded."`, y
> `Invoke-CoolifyApi.ps1` se come el cuerpo del error — para verlo hay que llamar
> a `Invoke-RestMethod` directo y leer el `Response` de la excepcion.
Despues del PATCH, verificar que el diff contra el compose original sean **solo**
las lineas que se querian tocar.
**Cambio de estado — requiere tu confirmacion antes de aplicarse.**
### Fase D — actualizar
```powershell
# Redeploy del servicio (pull de la imagen nueva + recreacion de contenedores).
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=c11xzy2tx2cdapm32f5b89vy"
```
Devuelve un `deployment_uuid`. Seguimiento:
```powershell
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments/<deployment_uuid>"
```
Al arrancar, el entrypoint corre `db:chatwoot_prepare` → `db:migrate` →
`ConfigLoader` (`reconcile_only_new: true`, no pisa nada). Esperar a que los 4
contenedores vuelvan a `healthy`:
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps | grep c11xzy2tx2cdapm32f5b89vy"
```
**Cambio de estado — requiere tu confirmacion.**
### Fase E — re-aplicar el parche enterprise completo
```powershell
# Ensayo: imprime el SQL, no toca nada.
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun
# Aplicar: 3 UPDATE + reactivacion de los 9 flags premium en todas las cuentas.
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
```
Criterio de exito (el script aborta si no se cumple):
- exactamente **3 lineas `UPDATE 1`**;
- `pendientes=ninguno` para cada cuenta;
- `self_hosted_enterprise=true`.
Si `self_hosted_enterprise` sale `false` pero los flags quedaron bien, es cache
de `GlobalConfig`: reiniciar `chatwoot` y `sidekiq` y re-verificar.
**Cambio de estado — requiere tu confirmacion.**
### Fase F — verificar
```powershell
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
```
Y a mano en `https://chat.urieljareth.org`:
1. Login como super admin.
2. `/super_admin/settings`: plan **Enterprise**, cantidad **10000**.
**NO pulsar `Refresh`** — revierte todo al instante.
3. Una funcion premium por cuenta (audit logs, SLA, custom roles o Captain).
4. Si habia branding propio, volverlo a poner (se reseteo, ver seccion 2).
### Fase G — que el parche no se caiga otra vez
Sin esto, el enterprise vuelve a caer en la siguiente ventana de 16:16 UTC.
#### Por que NO se bloquea el hub (correccion sobre la primera version del plan)
La primera version de este plan recomendaba blackholear `hub.2.chatwoot.com` con
`extra_hosts`, aprovechando el `return if @instance_info.blank?`. **Se descarto al
verificarlo contra este entorno**, por dos motivos:
1. **Rompe las notificaciones push del movil.** `hub.2.chatwoot.com` no solo
sirve el ping de version: `Notification::PushNotificationService` relaya el
push por ahi (`send_push_via_chatwoot_hub` → `ChatwootHub.send_push`) y ese
metodo corre **solo cuando Firebase no esta configurado**. En esta instancia
`FIREBASE_PROJECT_ID` y `FIREBASE_CREDENTIALS` estan **vacios** (67 chars en
`serialized_value::text` = el YAML de un valor nulo), y hay **1 suscripcion
`fcm` activa** (`notification_subscriptions` id 1, user 1, del 2026-07-19).
Bloquear el host le mata el push a ese usuario.
2. **Deja un job fallando todos los dias.** Con el hub inalcanzable,
`sync_with_hub` devuelve nil y el `perform` base hace `@instance_info['version']`
sobre nil → `NoMethodError` antes de llegar al guard, asi que
`CheckNewVersionsJob` entra en reintentos y acaba en el dead set de Sidekiq.
Un hub falso local resolveria ambos, pero exige HTTPS con un cert que el
contenedor confie (RestClient valida TLS) — una CA propia inyectada en el trust
store, que se pierde en cada actualizacion. No vale la pena.
#### Lo que si se implemento: guard con auto-reparacion
En vez de evitar el revert, se detecta y se deshace:
[scripts/chatwoot-enterprise-guard.sh](../../scripts/chatwoot-enterprise-guard.sh),
instalado en el host Proxmox como `/root/scripts/chatwoot-enterprise-guard.sh` con
`cron */5`.
Como funciona:
1. **Caso normal (barato):** 1 `SELECT` del plan y sale. **0.9 s**, sin escribir
en el log. 288 corridas al dia es ruido despreciable para el host.
2. **Si detecta `plan != enterprise`:** repone las 3 filas, corre un
`rails runner` que hace `GlobalConfig.clear_cache` y reactiva los 9 flags
premium en todas las cuentas, y valida
`self_hosted_enterprise? == true` + `pendientes=ninguno`. **12.7 s.**
3. Usa `flock` para no solaparse, y solo escribe en el log cuando actua.
Ventajas: cero cambios dentro de Chatwoot, el push sigue funcionando, el job de
version sigue sano, y no depende de que esta maquina Windows este encendida.
Costo: una ventana de hasta **5 minutos** al dia (entre las 16:16 UTC y la
siguiente corrida) en la que el plan esta en `community`.
Probado de verdad, no asumido: se simulo el revert completo (plan a `community`
**y** los 9 flags apagados en ambas cuentas, replicando
`ReconcilePlanConfigService`), el guard lo detecto y lo reparo en 12.7 s, la
segunda corrida fue no-op en 0.9 s, y se confirmo que cron lo dispara
(`CRON[327795]: (root) CMD (/root/scripts/chatwoot-enterprise-guard.sh)`).
#### Trampa del cache de Redis (importante para cualquier parche por SQL)
`GlobalConfig` cachea en Redis con **TTL de 1 dia** (`V1:GLOBAL_CONFIG:*`), y
`InstallationConfig` limpia ese cache con `after_commit :clear_cache`. Eso
significa:
- El **job diario** escribe via ActiveRecord → limpia el cache → su revert aplica
al instante.
- Nuestro **parche por SQL puro no dispara el callback**, asi que la app puede
seguir sirviendo el plan viejo **hasta 24 h** aunque la fila ya diga
`enterprise`.
Por eso tanto el guard como `Apply-ChatwootEnterprisePatch.ps1
-ReenableAccountFeatures` llaman explicitamente a `GlobalConfig.clear_cache`.
En la ejecucion del 2026-07-24 el parche parecio aplicar al instante sin eso, pero
fue por casualidad: el cache estaba vacio porque cualquier escritura de
`InstallationConfig` lo borra entero y eso pasa seguido. No hay que confiar en
esa casualidad.
### Fase H — rollback
Si la actualizacion rompe algo:
```powershell
# 1) Volver la imagen al digest anterior (compose en la UI de Coolify):
# image: 'chatwoot/chatwoot@sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175'
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=c11xzy2tx2cdapm32f5b89vy"
```
Si la migracion ya toco el esquema, la imagen vieja no va a arrancar contra la DB
nueva: hay que restaurar el dump de la Fase B **antes** de bajar la imagen.
```powershell
# 2) Restaurar el dump (DESTRUCTIVO: pisa la DB actual).
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct push 102 /root/backups/chatwoot-pre-4.16.1.dump /root/chatwoot-pre-4.16.1.dump"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp /root/chatwoot-pre-4.16.1.dump postgres-c11xzy2tx2cdapm32f5b89vy:/tmp/restore.dump"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD=\$POSTGRES_PASSWORD pg_restore -U \$POSTGRES_USER -d \$POSTGRES_DB --clean --if-exists /tmp/restore.dump'"
```
Luego Fase E otra vez.
## 4. Orden de ejecucion resumido
```
A pre-checks (lectura)
B backup DB + compose/.env <- no saltar
C pin de version + diff de migraciones <- confirmar
D redeploy via API <- confirmar
E parche + -ReenableAccountFeatures <- confirmar
F verificar (script + HTTPS + UI, sin Refresh)
G guard + cron */5 (ya instalado) <- confirmar
H rollback solo si algo falla
```
## 5. Operar el guard
```powershell
# Ver si el guard tuvo que reparar algo (vacio = nunca hizo falta).
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "cat /var/log/chatwoot-enterprise-guard.log"
# Forzar una corrida.
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "/root/scripts/chatwoot-enterprise-guard.sh; echo rc=\$?"
# Confirmar que cron lo dispara.
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "journalctl --since '-20 min' --no-pager | grep chatwoot-enterprise-guard"
```
Lo normal es un log **vacio o con pocas lineas**. Un `REPARADO` por dia es lo
esperado (el revert de las 16:16 UTC). Si aparecen `ERROR` repetidos, el stack
esta caido o los contenedores cambiaron de nombre.
**Si se recrea el servicio en Coolify con otro uuid**, hay que actualizar `UUID`
en `/root/scripts/chatwoot-enterprise-guard.sh` — el guard lo tiene hardcodeado.
## 6. Notas de riesgo
- **PostgreSQL 12.19** y la imagen `pgvector/pgvector:pg12` tiene ~23 meses.
Un salto de version mayor de Chatwoot probablemente exija PG >= 13. Antes de
actualizar mas alla de 4.16.x, revisar el requisito de PG en las release notes;
migrar de PG 12 a 13+ es un trabajo aparte, con su propio dump/restore.
- **`Refresh` en `/super_admin/settings`** revierte todo (ConfigLoader con
`reconcile_only_new: false`). El guard lo repararia en <= 5 min, pero conviene
no pulsarlo.
- Ahora que la imagen esta **pineada a `v4.16.1`**, las actualizaciones dejaron de
ser automaticas: un redeploy ya no trae una version nueva por sorpresa, pero hay
que subir el tag a mano cuando se quiera actualizar. Ese es el punto del pin.
- El `.env` del servicio y el `.tgz` del backup **contienen secretos**: se
quedan en el host Proxmox. Nunca copiarlos al repo.
- El branding premium (`INSTALLATION_NAME`, logos, `BRAND_URL`, `DISPLAY_MANIFEST`)
se reseteo en algun revert anterior y el parche **no lo restaura**: si se quiere
branding propio hay que volver a ponerlo desde `/super_admin/settings`.
- Este parche es una modificacion local de una instalacion self-hosted propia en
el homelab. No se distribuye ni se revende.
## 7. Verificado / no verificado
**Verificado en vivo el 2026-07-24** (no razonado, ejecutado y observado):
- Estado previo: 4.16.0, plan `community`, cantidad `0`,
`self_hosted_enterprise? = false`, los 9 flags premium apagados en ambas cuentas.
- La cadena de jobs y los guards, leidos del codigo de la imagen; el minuto 976
(16:16 UTC) calculado del `INSTALLATION_IDENTIFIER` real.
- Backup: dump de 667K con 97 tablas (`pg_restore -l`) + sha256.
- Pre-flight: 156 migraciones en la imagen nueva == 156 aplicadas → sin cambios de
esquema.
- `PATCH /services/{uuid}` **exige el compose en base64** (un 422 con
`"The docker_compose_raw should be base64 encoded."` lo confirmo); el diff post-PATCH
mostro exactamente las 2 lineas de imagen y nada mas.
- Deploy: 4.16.1 corriendo, 4/4 `healthy`, ~2 min de corte.
- Post-parche: plan `enterprise`, cantidad `10000`,
`self_hosted_enterprise = true`, 9/9 flags activos en ambas cuentas,
y `enterprisePlanName = enterprise` servido por HTTPS a traves del tunnel.
- Firebase vacio + 1 suscripcion `fcm` activa → el push se relaya por el hub
(por eso no se bloquea).
- `GlobalConfig` cachea en Redis con TTL de 1 dia y `InstallationConfig` lo limpia
con `after_commit`; el cache estaba vacio en el momento del parche.
- El guard: no-op en 0.9 s, reparacion completa en 12.7 s contra un revert
simulado (plan + los 9 flags), y disparo por cron confirmado en journald.
**No verificado:**
- Que 4.16.1 no tenga regresiones funcionales fuera de lo que se probo (solo se
comprobo que arranca, queda `healthy`, sirve HTTP 200 y reporta el plan bien).
- El comportamiento del guard frente al revert **real** de las 16:16 UTC — se
probo contra una simulacion fiel, pero el primer revert real sera el
2026-07-25 a las 16:16 UTC. Revisar el log ese dia.
- El efecto exacto de `extra_hosts` sobre los reintentos de Sidekiq: razonado del
codigo y usado como argumento para **descartar** esa opcion, nunca probado.
- Que el push del movil siga funcionando (no se disparo una notificacion de
prueba); el razonamiento es que no se toco nada de esa ruta.
---
## Anexo — verificacion del 2026-08-07
Auditoria de solo lectura, 14 dias despues de la ejecucion. Tres cosas cambiaron.
### 1. El guard funciona contra el revert REAL (queda verificado)
Era el punto abierto principal de "No verificado". El log
`/var/log/chatwoot-enterprise-guard.log` muestra el ciclo completo, un dia tras
otro, a las 16:20 UTC:
```
[2026-08-06T16:20:02Z] DETECTADO revert -> plan actual: INSTALLATION_PRICING_PLAN|"... value: community ..." . Reparando...
[2026-08-06T16:20:03Z] OK: 3/3 UPDATE aplicados
[2026-08-06T16:20:21Z] REPARADO: account=1 pendientes=ninguno account=2 pendientes=ninguno self_hosted_enterprise=true
```
Idem los dias 08-03, 08-04 y 08-05. El revert diario ocurre de verdad y el guard
lo deshace en ~20 s. Deja de ser una hipotesis.
### 2. El pin de version NO sostuvo: corre `v4.16.2`
```
chatwoot-c11xzy2tx2cdapm32f5b89vy chatwoot/chatwoot:v4.16.2
sidekiq-c11xzy2tx2cdapm32f5b89vy chatwoot/chatwoot:v4.16.2
```
La imagen se habia pineado a `v4.16.1` via API. Hoy corre `v4.16.2`, asi que en
algun momento entre el 2026-07-24 y hoy alguien o algo movio el tag y hubo un
redeploy. **Antes de asumir que el pin protege, verificalo:**
```powershell
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}|{{.Image}}' | grep chatwoot"
```
### 3. El guard fallo durante la ventana del update
Cuatro errores el 2026-08-07, los primeros dos en la ventana del revert diario:
```
[2026-08-07T16:17:15Z] ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)
[2026-08-07T16:20:03Z] ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)
[2026-08-07T23:11:45Z] ERROR: ...
[2026-08-07T23:15:02Z] ERROR: ...
```
**Estado tras la auditoria: sano.** Una corrida manual con `bash -x` lee el plan
sin problema y sale 0, y el valor en la DB es `enterprise`:
```
INSTALLATION_PRICING_PLAN|"--- !ruby/hash:...\nvalue: enterprise\n"
```
Es decir: los errores fueron transitorios, coincidentes con la recreacion de
contenedores del update a `v4.16.2`. **No hay accion urgente.** Pero deja
expuesto un riesgo estructural.
### 4. Riesgo estructural: el guard tiene el nombre del contenedor hardcodeado
`scripts/chatwoot-enterprise-guard.sh` fija:
```bash
UUID=c11xzy2tx2cdapm32f5b89vy
DB_CT="postgres-$UUID"
APP_CT="chatwoot-$UUID"
```
Mientras el uuid del *service* no cambie, los nombres se mantienen — y en este
caso se mantuvieron. Pero el propio mensaje de error del guard nombra la causa
("contenedor renombrado?"), y **un redeploy que cambie el sufijo lo deja ciego
sin avisar**: el guard sale con codigo 1 y solo escribe una linea en un log que
nadie lee. El revert diario dejaria de repararse en silencio.
Mitigacion pendiente (no aplicada — requiere confirmacion porque toca el host):
hacer que el guard **resuelva el nombre por patron** en vez de fijarlo, p. ej.
`docker ps --format '{{.Names}}' | grep -m1 '^postgres-'` acotado al uuid del
service consultado a la API de Coolify. Y que N fallos consecutivos escalen a
algo visible, no solo al log.
### Comprobacion rapida del estado
```powershell
. .\.env.local.ps1
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "tail -20 /var/log/chatwoot-enterprise-guard.log"
```
Nota sobre el quoting: cualquier lectura directa de la DB necesita el patron
base64, porque `Invoke-ProxmoxSsh.ps1` corrompe las comillas anidadas. Ver
[../TOOL-INDEX.md](../TOOL-INDEX.md) §1.2.
+5 -5
View File
@@ -20,9 +20,9 @@ Síntomas típicos que llevan aquí:
- Dominio de app con `502 Bad Gateway` o `530` - Dominio de app con `502 Bad Gateway` o `530`
Documentos relacionados: Documentos relacionados:
[ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md](../ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md) · [2026-04-11-cloudflare-tunnel-websocket-tls.md](../incidentes/2026-04-11-cloudflare-tunnel-websocket-tls.md) ·
[cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md) · [guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md) ·
[issue-coolify-static-app-deploy.md](../issue-coolify-static-app-deploy.md) [2026-04-11-coolify-static-app-deploy.md](../incidentes/2026-04-11-coolify-static-app-deploy.md)
--- ---
@@ -162,7 +162,7 @@ $body = @{
> ⚠️ El último elemento `http_status:404` (sin hostname) es obligatorio o la API > ⚠️ El último elemento `http_status:404` (sin hostname) es obligatorio o la API
> rechaza la configuración. La referencia completa de endpoints (DNS, crear túnel, > rechaza la configuración. La referencia completa de endpoints (DNS, crear túnel,
> obtener token) está en [cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md). > obtener token) está en [guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md).
**Cualquier `PUT`/`POST` a Cloudflare es un cambio de estado → confirmar con el usuario y **Cualquier `PUT`/`POST` a Cloudflare es un cambio de estado → confirmar con el usuario y
capturar el estado actual (paso 2.2) antes de aplicar, para tener rollback.** capturar el estado actual (paso 2.2) antes de aplicar, para tener rollback.**
@@ -187,7 +187,7 @@ terminal en cualquier contenedor. Debe conectar sin `Terminal websocket connecti
cambio son **residuales** de conexiones ya abiertas; desaparecen solos. cambio son **residuales** de conexiones ya abiertas; desaparecen solos.
- Si el túnel se cae cada ~5 min con `failed to dial to edge with quic: timeout`, - Si el túnel se cae cada ~5 min con `failed to dial to edge with quic: timeout`,
el contenedor debe correr con `--protocol http2` (UDP/7844 suele estar bloqueado el contenedor debe correr con `--protocol http2` (UDP/7844 suele estar bloqueado
en redes domésticas). Ver PASO 6 de [cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md). en redes domésticas). Ver PASO 6 de [guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md).
- `cloudflared` es distroless: para leer archivos internos usa `docker cp`, no `cat` directo. - `cloudflared` es distroless: para leer archivos internos usa `docker cp`, no `cat` directo.
--- ---
+1 -1
View File
@@ -8,7 +8,7 @@ Usa un archivo privado `.env.local.ps1` con estos valores:
$env:PROXMOX_HOST = "192.168.0.200" $env:PROXMOX_HOST = "192.168.0.200"
$env:PROXMOX_NODE = "thinkcentre" $env:PROXMOX_NODE = "thinkcentre"
$env:PROXMOX_USER = "root" $env:PROXMOX_USER = "root"
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" $env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519"
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json" $env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw" $env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET" $env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
+1 -1
View File
@@ -36,6 +36,6 @@ Requiere confirmacion explicita.
Usar solo cuando haga falta inspeccion manual. Usar solo cuando haga falta inspeccion manual.
```powershell ```powershell
ssh -i "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" root@192.168.0.200 ssh -i "keys\proxmox_ed25519" root@192.168.0.200
pct exec 102 -- docker exec -it <container> /bin/sh pct exec 102 -- docker exec -it <container> /bin/sh
``` ```
+1 -1
View File
@@ -23,7 +23,7 @@ Si el endpoint API de Coolify no esta disponible o no hay token cargado:
No usar `sed` para editar `config.php`. Usar el script PHP del repo: No usar `sed` para editar `config.php`. Usar el script PHP del repo:
```powershell ```powershell
scp -o StrictHostKeyChecking=no -i "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" .\scripts\fix-nextcloud-config.php root@192.168.0.200:/tmp/fix-nextcloud-config.php scp -o StrictHostKeyChecking=no -i "keys\proxmox_ed25519" .\scripts\fix-nextcloud-config.php root@192.168.0.200:/tmp/fix-nextcloud-config.php
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct push 102 /tmp/fix-nextcloud-config.php /tmp/fix-nextcloud-config.php" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct push 102 /tmp/fix-nextcloud-config.php /tmp/fix-nextcloud-config.php"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp /tmp/fix-nextcloud-config.php nextcloud-hdcdpkm0jko3qqvn5683ercc:/tmp/fix-nextcloud-config.php" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp /tmp/fix-nextcloud-config.php nextcloud-hdcdpkm0jko3qqvn5683ercc:/tmp/fix-nextcloud-config.php"
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec nextcloud-hdcdpkm0jko3qqvn5683ercc php /tmp/fix-nextcloud-config.php /config/www/nextcloud/config/config.php nextcloudsuite.urieljareth.org nextcloud-db" .\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec nextcloud-hdcdpkm0jko3qqvn5683ercc php /tmp/fix-nextcloud-config.php /config/www/nextcloud/config/config.php nextcloudsuite.urieljareth.org nextcloud-db"
+7 -3
View File
@@ -1,6 +1,6 @@
--- ---
name: gitea-agent name: gitea-agent
description: Operate the local self-hosted Gitea instance (gitea-...urieljareth.org) for this Proxmox and Coolify Manager project. Use when Codex needs to create/list/search/mirror Gitea repositories, push a local project to Gitea headlessly (token in extraHeader, no wincred), inspect branches/releases/hooks, or wire a `gitea` remote alongside a GitHub `origin`. Distinct from coolify-deploy (which ships to Coolify) — this skill only manages the git hosting layer on Gitea. description: Operate the local self-hosted Gitea instance (gitea-...urieljareth.org) for this Proxmox and Coolify Manager project. Use when the agent needs to create/list/search/mirror Gitea repositories, push a local project to Gitea headlessly (token in extraHeader, no wincred), inspect branches/releases/hooks, or wire a `gitea` remote alongside a GitHub `origin`. Distinct from coolify-deploy (which ships to Coolify) — this skill only manages the git hosting layer on Gitea.
--- ---
# Gitea Agent # Gitea Agent
@@ -13,8 +13,12 @@ only via its public URL.
1. Load private values: `. .\.env.local.ps1` (provides `GITEA_URL`, `GITEA_USER`, 1. Load private values: `. .\.env.local.ps1` (provides `GITEA_URL`, `GITEA_USER`,
`GITEA_TOKEN`). `GITEA_TOKEN`).
2. Run the smoke test: `.\gitea_skill\scripts\Test-GiteaConnection.ps1`. 2. Skim [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) §5 for the verified script
3. Only after it passes, perform the requested operation. signatures. Note that in this skill's wrappers `-Raw` returns
`{Status, Headers, Body}` and the **default returns objects** — the opposite of
the Coolify and Cloudflare wrappers (§1.1).
3. Run the smoke test: `.\gitea_skill\scripts\Test-GiteaConnection.ps1`.
4. Only after it passes, perform the requested operation.
## Local context (verified 2026-07-19) ## Local context (verified 2026-07-19)
+3 -1
View File
@@ -78,7 +78,9 @@ try {
$bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" } $bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" }
$statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim() $statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim()
if ($statusLine -notmatch " 20[04-9] ") { # 2xx completo: la creación de repos/hooks/releases responde 201 y la clase
# de caracteres anterior ([04-9]) lo trataba como error pese al éxito.
if ($statusLine -notmatch " 20[0-9] ") {
$snippet = $bodyBlock.Trim() $snippet = $bodyBlock.Trim()
if ($snippet.Length -gt 500) { $snippet = $snippet.Substring(0, 500) + "..." } if ($snippet.Length -gt 500) { $snippet = $snippet.Substring(0, 500) + "..." }
throw "Gitea API HTTP error: $statusLine`nURI: $uri`nBody: $snippet" throw "Gitea API HTTP error: $statusLine`nURI: $uri`nBody: $snippet"
-3
View File
@@ -1,3 +0,0 @@
# Migrado
Usa `../../agent/SKILL.md`.
-3
View File
@@ -1,3 +0,0 @@
# Migrado
Usa `../../agent/TOOLS.md`.
-6
View File
@@ -1,6 +0,0 @@
# Migrado
El historial util de incidentes esta consolidado en:
- `../../docs/runbooks/incidentes-openclaw.md`
- `../../docs/runbooks/coolify-docker.md`
-3
View File
@@ -1,3 +0,0 @@
# Migrado
Usa `../../agent/SKILL.md`.
-3
View File
@@ -1,3 +0,0 @@
# Migrado
Usa `../../agent/TOOLS.md`.
+134 -5
View File
@@ -8,18 +8,28 @@
# Criterio de exito: psql imprime 3 lineas "UPDATE 1" (una por sentencia). # Criterio de exito: psql imprime 3 lineas "UPDATE 1" (una por sentencia).
# Tras aplicar, NO pulsar "Refresh" en /super_admin/settings. # Tras aplicar, NO pulsar "Refresh" en /super_admin/settings.
# #
# IMPORTANTE: los 3 UPDATE por si solos NO alcanzan cuando el plan ya se habia
# revertido a 'community'. Al revertirse, Internal::ReconcilePlanConfigService
# apaga los 9 feature flags premium en CADA cuenta, y eso vive en la tabla
# accounts (bitmask), no en installation_configs. Usa -ReenableAccountFeatures
# para reactivarlos. Ver docs/runbooks/chatwoot-update.md.
#
# Uso: # Uso:
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 # .\scripts\Apply-ChatwootEnterprisePatch.ps1
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun # .\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy" # .\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy"
[CmdletBinding()] [CmdletBinding()]
param( param(
[switch]$DryRun, [switch]$DryRun,
[switch]$ReenableAccountFeatures,
[string]$ServiceUuid = "c11xzy2tx2cdapm32f5b89vy",
[string]$Container = "", [string]$Container = "",
[string]$AppContainer = "",
[string]$LxcId = "102", [string]$LxcId = "102",
[string]$ProxmoxHost = $(if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" }), [string]$ProxmoxHost = $(if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" }),
[string]$SshKey = $(if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" }) [string]$SshKey = $(if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { Join-Path $PSScriptRoot "..\keys\proxmox_ed25519" })
) )
# Encoding UTF-8 sin BOM (BOM rompe el shebang #!/bin/bash en Linux). # Encoding UTF-8 sin BOM (BOM rompe el shebang #!/bin/bash en Linux).
@@ -44,13 +54,18 @@ function Invoke-Remote {
return & ssh @args return & ssh @args
} }
if (-not $AppContainer) { $AppContainer = "chatwoot-$ServiceUuid" }
# --- 1) Resolver contenedor Postgres de Chatwoot -------------------------- # --- 1) Resolver contenedor Postgres de Chatwoot --------------------------
if (-not $Container) { if (-not $Container) {
# Dos greps encadenados en vez de un solo patron: Coolify nombra los
# contenedores <servicio>-<uuid> (postgres-c11xzy...), asi que un patron
# "<uuid>.*postgres" nunca casa. El orden no importa con greps separados.
$detect = @' $detect = @'
#!/bin/bash #!/bin/bash
pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -Ei "c11xzy2tx2cdapm32f5b89vy.*(pgvector|postgres|db)" | head -n1' pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -F "__UUID__" | grep -Ei "(pgvector|postgres|db)" | head -n1'
'@ '@
$detect = $detect.Replace('__LXC__', $LxcId) $detect = $detect.Replace('__LXC__', $LxcId).Replace('__UUID__', $ServiceUuid)
$tmpDetect = [IO.Path]::GetTempFileName() + ".sh" $tmpDetect = [IO.Path]::GetTempFileName() + ".sh"
[IO.File]::WriteAllText($tmpDetect, $detect, $utf8NoBom) [IO.File]::WriteAllText($tmpDetect, $detect, $utf8NoBom)
@@ -72,8 +87,9 @@ pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -Ei "c11xzy
if ($LASTEXITCODE -ne 0) { throw "Listado remoto fallo (exit $LASTEXITCODE)." } if ($LASTEXITCODE -ne 0) { throw "Listado remoto fallo (exit $LASTEXITCODE)." }
$Container = @($cand | Where-Object { $_ -match '\S' })[0] $Container = @($cand | Where-Object { $_ -match '\S' })[0]
if (-not $Container) { if (-not $Container) {
throw "No se encontro contenedor. Pasa -Container explicito (ej: postgres-c11xzy2tx2cdapm32f5b89vy)." throw "No se encontro contenedor Postgres para el servicio $ServiceUuid en el LXC $LxcId. Pasa -Container explicito (ej: postgres-$ServiceUuid)."
} }
$Container = $Container.Trim()
} finally { } finally {
Remove-Item -LiteralPath $tmpDetect -ErrorAction SilentlyContinue Remove-Item -LiteralPath $tmpDetect -ErrorAction SilentlyContinue
} }
@@ -174,8 +190,23 @@ if ($DryRun) {
Write-Host "------" Write-Host "------"
Write-Host "[dry-run] apply.sh:" Write-Host "[dry-run] apply.sh:"
Write-Host "------" Write-Host "------"
Write-Host $applyScript # PGPASSWORD se enmascara: la regla del repo es que ningun secreto salga por
# stdout ni quede en un log. (String.Replace revienta con un patron vacio,
# de ahi el guard.)
$safeApply = if ([string]::IsNullOrEmpty($pgPass)) { $applyScript } else { $applyScript.Replace($pgPass, "********") }
Write-Host $safeApply
Write-Host "------" Write-Host "------"
if ($ReenableAccountFeatures) {
Write-Host "[dry-run] Ademas reactivaria estos feature flags premium en TODAS las cuentas,"
Write-Host " via 'rails runner' en $AppContainer :"
Write-Host " disable_branding audit_logs sla custom_roles captain_integration"
Write-Host " captain_integration_v2 captain_document_auto_sync csat_review_notes"
Write-Host " conversation_required_attributes"
}
else {
Write-Host "[dry-run] Los feature flags premium por cuenta NO se tocarian."
Write-Host " Agrega -ReenableAccountFeatures si el plan venia de 'community'."
}
return return
} }
@@ -266,3 +297,101 @@ Invoke-Remote "rm -f $remoteSql $remoteApply $remoteVerify" | Out-Null
Write-Host "" Write-Host ""
Write-Host "[OK] Parche enterprise aplicado correctamente (3/3 UPDATE 1)." Write-Host "[OK] Parche enterprise aplicado correctamente (3/3 UPDATE 1)."
Write-Host " NO pulsar 'Refresh' en /super_admin/settings." Write-Host " NO pulsar 'Refresh' en /super_admin/settings."
# --- 6) Reactivar feature flags premium por cuenta -------------------------
# Cuando el plan se revierte a 'community', Internal::ReconcilePlanConfigService
# corre account.disable_features!(*premium_features) sobre TODAS las cuentas.
# Esos flags viven en accounts.feature_flags (bitmask) y los 3 UPDATE de arriba
# no los tocan: hay que reactivarlos explicitamente o la UI sigue sin enterprise.
if (-not $ReenableAccountFeatures) {
Write-Host ""
Write-Host "[!] Los feature flags premium por cuenta NO se tocaron."
Write-Host " Si el plan venia de 'community', vuelve a correr con -ReenableAccountFeatures."
return
}
Write-Host ""
Write-Host "[*] Reactivando feature flags premium por cuenta (arranca Rails, ~40 s) ..."
$ruby = @'
PREMIUM = %w[
disable_branding audit_logs sla custom_roles
captain_integration captain_integration_v2 captain_document_auto_sync
csat_review_notes conversation_required_attributes
]
# Los 3 UPDATE se hacen por SQL puro, asi que NO disparan el
# `after_commit :clear_cache` de InstallationConfig. GlobalConfig cachea en Redis
# con TTL de 1 dia (V1:GLOBAL_CONFIG:*), asi que sin esta limpieza la app puede
# seguir sirviendo el plan viejo hasta 24 h.
GlobalConfig.clear_cache
puts "global_config_cache=limpiado"
Account.find_each do |account|
before = PREMIUM.reject { |f| account.feature_enabled?(f) }
account.enable_features!(*PREMIUM)
account.reload
after = PREMIUM.reject { |f| account.feature_enabled?(f) }
puts "account=#{account.id}|#{account.name}|reactivados=#{before.empty? ? 'ninguno' : before.join(',')}|pendientes=#{after.empty? ? 'ninguno' : after.join(',')}"
end
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
'@
$tmpRb = [IO.Path]::GetTempFileName()
$remoteRb = "/tmp/chatwoot-reenable-features.rb"
$tmpRunner = [IO.Path]::GetTempFileName()
$remoteRunner = "/tmp/chatwoot-reenable-features.sh"
$runner = @'
set -e
pct push __LXC__ __RB__ __RB__
pct exec __LXC__ -- docker cp __RB__ __APP__:__RB__
pct exec __LXC__ -- docker exec -i __APP__ bundle exec rails runner __RB__
pct exec __LXC__ -- docker exec -i __APP__ rm -f __RB__
pct exec __LXC__ -- rm -f __RB__
'@
$runner = $runner.Replace('__LXC__', $LxcId).Replace('__APP__', $AppContainer).Replace('__RB__', $remoteRb)
try {
[IO.File]::WriteAllText($tmpRb, $ruby, $utf8NoBom)
[IO.File]::WriteAllText($tmpRunner, $runner, $utf8NoBom)
foreach ($pair in @(@($tmpRb, $remoteRb), @($tmpRunner, $remoteRunner))) {
$scpArgs = @(
"-o", "BatchMode=yes"
"-o", "ConnectTimeout=15"
"-o", "StrictHostKeyChecking=no"
"-i", $SshKey
$pair[0]
"root@${ProxmoxHost}:$($pair[1])"
)
& scp @scpArgs | Out-Null
if ($LASTEXITCODE -ne 0) { throw "scp de $($pair[1]) fallo (exit $LASTEXITCODE)." }
}
$featOut = Invoke-Remote "bash $remoteRunner"
$featExit = $LASTEXITCODE
$featOut | ForEach-Object { Write-Host $_ }
}
finally {
Remove-Item -LiteralPath $tmpRb, $tmpRunner -ErrorAction SilentlyContinue
Invoke-Remote "rm -f $remoteRb $remoteRunner" | Out-Null
}
if ($featExit -ne 0) {
throw "La reactivacion de feature flags fallo (exit $featExit)."
}
$pending = @($featOut | Where-Object { $_ -match 'pendientes=(?!ninguno)' })
if ($pending.Count -gt 0) {
throw "Quedaron feature flags premium sin activar en $($pending.Count) cuenta(s). Revisa la salida de arriba."
}
$selfHosted = @($featOut | Where-Object { $_ -like "self_hosted_enterprise=*" })[0]
Write-Host ""
if ($selfHosted -eq "self_hosted_enterprise=true") {
Write-Host "[OK] Enterprise activo: plan=enterprise y feature flags premium reactivados en todas las cuentas."
}
else {
Write-Host "[WARN] Feature flags reactivados, pero ChatwootApp.self_hosted_enterprise? no dio true ($selfHosted)."
Write-Host " Reinicia el stack para limpiar el cache de GlobalConfig y vuelve a verificar con:"
Write-Host " .\scripts\Get-ChatwootLicenseStatus.ps1 -Deep"
}
+250
View File
@@ -0,0 +1,250 @@
# Reporta el estado de la licencia enterprise de Chatwoot en Coolify (LXC 102).
#
# Solo lectura. Pensado como pre-check y post-check del runbook de actualizacion
# (docs/runbooks/chatwoot-update.md).
#
# Que reporta:
# - version instalada vs ultima conocida por el hub
# - las 3 filas de public.installation_configs del parche enterprise
# - con -Deep: ChatwootApp.self_hosted_enterprise? y los 9 feature flags
# premium por cuenta (esto tarda ~40 s porque arranca Rails)
#
# Uso:
# .\scripts\Get-ChatwootLicenseStatus.ps1
# .\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
[CmdletBinding()]
param(
[string]$ServiceUuid = "c11xzy2tx2cdapm32f5b89vy",
[string]$AppContainer = "",
[string]$DbContainer = "",
[switch]$Deep
)
$ErrorActionPreference = "Stop"
. (Join-Path $PSScriptRoot "ProxmoxAgent.ps1")
$config = Assert-ProxmoxConfig
$lxc = $config.CoolifyLxc
$utf8NoBom = New-Object System.Text.UTF8Encoding($false)
# Feature flags premium que Internal::ReconcilePlanConfigService apaga cuando el
# plan vuelve a 'community' (enterprise/config/premium_features.yml).
$premiumFeatures = @(
"disable_branding", "audit_logs", "sla", "custom_roles",
"captain_integration", "captain_integration_v2", "captain_document_auto_sync",
"csat_review_notes", "conversation_required_attributes"
)
function Invoke-Remote {
param([string]$RemoteCmd)
$sshArgs = @(
"-o", "BatchMode=yes"
"-o", "ConnectTimeout=20"
"-o", "StrictHostKeyChecking=no"
"-i", $config.SshKey
"$($config.User)@$($config.HostName)"
$RemoteCmd
)
return & ssh @sshArgs
}
# Sube un script como archivo y lo ejecuta con bash. Evita el infierno de
# escaping de comillas sobre SSH (ver docs/casos/chatwoot-enterprise-patch.md).
function Invoke-RemoteScript {
param(
[string]$Body,
[string]$RemoteName
)
$tmp = [IO.Path]::GetTempFileName()
try {
[IO.File]::WriteAllText($tmp, $Body, $utf8NoBom)
$scpArgs = @(
"-o", "BatchMode=yes"
"-o", "ConnectTimeout=20"
"-o", "StrictHostKeyChecking=no"
"-i", $config.SshKey
$tmp
"$($config.User)@$($config.HostName):/tmp/$RemoteName"
)
& scp @scpArgs | Out-Null
if ($LASTEXITCODE -ne 0) { throw "scp de $RemoteName fallo (exit $LASTEXITCODE)." }
$out = Invoke-Remote "bash /tmp/$RemoteName"
$code = $LASTEXITCODE
Invoke-Remote "rm -f /tmp/$RemoteName" | Out-Null
if ($code -ne 0) { throw "$RemoteName fallo en el host (exit $code)." }
return $out
}
finally {
Remove-Item -LiteralPath $tmp -ErrorAction SilentlyContinue
}
}
# --- 1) Resolver contenedores del stack -----------------------------------
if (-not $AppContainer) { $AppContainer = "chatwoot-$ServiceUuid" }
if (-not $DbContainer) { $DbContainer = "postgres-$ServiceUuid" }
$psScript = @'
pct exec __LXC__ -- docker ps --format "{{.Names}} {{.Status}}" | grep -E "__UUID__"
'@
$psScript = $psScript.Replace('__LXC__', $lxc).Replace('__UUID__', $ServiceUuid)
$containers = Invoke-RemoteScript -Body $psScript -RemoteName "cw-status-ps.sh"
Write-Host "=== Stack Chatwoot (LXC $lxc) ==="
if (-not $containers) {
throw "No se encontro ningun contenedor con el UUID $ServiceUuid en el LXC $lxc."
}
$containers | ForEach-Object { Write-Host " $_" }
# --- 2) Version instalada -------------------------------------------------
$verScript = @'
pct exec __LXC__ -- docker exec -i __APP__ sh -c 'grep -m1 "version:" /app/config/app.yml'
'@
$verScript = $verScript.Replace('__LXC__', $lxc).Replace('__APP__', $AppContainer)
$verRaw = (Invoke-RemoteScript -Body $verScript -RemoteName "cw-status-ver.sh") -join " "
$version = if ($verRaw -match "'([^']+)'") { $Matches[1] } else { $verRaw.Trim() }
Write-Host ""
Write-Host "=== Version ==="
Write-Host " instalada: $version"
# --- 3) Filas del parche enterprise ---------------------------------------
# La query se ejecuta dentro del contenedor Postgres para que POSTGRES_USER /
# POSTGRES_PASSWORD nunca salgan del contenedor ni queden en logs.
$sqlScript = @'
pct exec __LXC__ -- docker exec -i __DB__ bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -F "|" -c "SELECT name, serialized_value FROM public.installation_configs ORDER BY name"'
'@
$sqlScript = $sqlScript.Replace('__LXC__', $lxc).Replace('__DB__', $DbContainer)
$rows = Invoke-RemoteScript -Body $sqlScript -RemoteName "cw-status-sql.sh"
function Get-ConfigValue {
param([string]$Name)
$line = @($rows | Where-Object { $_ -like "$Name|*" })[0]
if (-not $line) { return "<ausente>" }
# serialized_value es YAML de Ruby: "--- ...\nvalue: X\n"
if ($line -match 'value:\s*([^\\"]*)') { return $Matches[1].Trim() }
return $line
}
$plan = Get-ConfigValue "INSTALLATION_PRICING_PLAN"
$quantity = Get-ConfigValue "INSTALLATION_PRICING_PLAN_QUANTITY"
$identifier = Get-ConfigValue "INSTALLATION_IDENTIFIER"
$instName = Get-ConfigValue "INSTALLATION_NAME"
Write-Host ""
Write-Host "=== installation_configs (parche enterprise) ==="
Write-Host " INSTALLATION_PRICING_PLAN = $plan"
Write-Host " INSTALLATION_PRICING_PLAN_QUANTITY = $quantity"
Write-Host " INSTALLATION_IDENTIFIER = $identifier"
Write-Host " INSTALLATION_NAME = $instName"
# La ventana diaria de revert es determinista:
# Internal::TriggerDailyScheduledItemsJob (cron 0 0 * * *) programa
# Internal::CheckNewVersionsJob en beginning_of_day + (MD5(identifier).hex % 1440) minutos.
if ($identifier -and $identifier -ne "<ausente>") {
$md5 = [System.Security.Cryptography.MD5]::Create()
try {
$digest = ($md5.ComputeHash([Text.Encoding]::UTF8.GetBytes($identifier)) |
ForEach-Object { $_.ToString("x2") }) -join ""
$asInt = [Numerics.BigInteger]::Parse("0$digest", "AllowHexSpecifier")
$minute = [int]($asInt % 1440)
Write-Host (" ventana diaria de revert = {0:d2}:{1:d2} UTC (minuto {2})" -f [int][math]::Floor($minute / 60), [int]($minute % 60), $minute)
}
finally {
$md5.Dispose()
}
}
$planOk = ($plan -eq "enterprise")
# --- 4) Chequeo profundo con Rails ----------------------------------------
$featuresOk = $null
if ($Deep) {
Write-Host ""
Write-Host "[*] Arrancando Rails para el chequeo profundo (~40 s) ..."
$ruby = @'
PREMIUM = %w[__FEATURES__]
puts "version=#{Chatwoot.config[:version]}"
puts "enterprise=#{ChatwootApp.enterprise?}"
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
puts "latest_known_version=#{Redis::Alfred.get(Redis::Alfred::LATEST_CHATWOOT_VERSION)}"
Account.find_each do |a|
off = PREMIUM.reject { |f| a.feature_enabled?(f) }
puts "account=#{a.id}|#{a.name}|#{off.empty? ? 'OK' : off.join(',')}"
end
'@
$ruby = $ruby.Replace('__FEATURES__', ($premiumFeatures -join " "))
$tmpRb = [IO.Path]::GetTempFileName()
try {
[IO.File]::WriteAllText($tmpRb, $ruby, $utf8NoBom)
$scpArgs = @(
"-o", "BatchMode=yes"
"-o", "ConnectTimeout=20"
"-o", "StrictHostKeyChecking=no"
"-i", $config.SshKey
$tmpRb
"$($config.User)@$($config.HostName):/tmp/cw-deep.rb"
)
& scp @scpArgs | Out-Null
if ($LASTEXITCODE -ne 0) { throw "scp del script Ruby fallo (exit $LASTEXITCODE)." }
# El .rb tiene que llegar al filesystem del contenedor: host -> LXC -> docker.
$runner = @'
set -e
pct push __LXC__ /tmp/cw-deep.rb /tmp/cw-deep.rb
pct exec __LXC__ -- docker cp /tmp/cw-deep.rb __APP__:/tmp/cw-deep.rb
pct exec __LXC__ -- docker exec -i __APP__ bundle exec rails runner /tmp/cw-deep.rb
pct exec __LXC__ -- docker exec -i __APP__ rm -f /tmp/cw-deep.rb
pct exec __LXC__ -- rm -f /tmp/cw-deep.rb
'@
$runner = $runner.Replace('__LXC__', $lxc).Replace('__APP__', $AppContainer)
$deepOut = Invoke-RemoteScript -Body $runner -RemoteName "cw-status-deep.sh"
Invoke-Remote "rm -f /tmp/cw-deep.rb" | Out-Null
}
finally {
Remove-Item -LiteralPath $tmpRb -ErrorAction SilentlyContinue
}
$selfHosted = @($deepOut | Where-Object { $_ -like "self_hosted_enterprise=*" })[0]
$latest = @($deepOut | Where-Object { $_ -like "latest_known_version=*" })[0]
$accounts = @($deepOut | Where-Object { $_ -like "account=*" })
Write-Host ""
Write-Host "=== Rails ==="
if ($latest) { Write-Host " $latest" }
if ($selfHosted) { Write-Host " $selfHosted" }
Write-Host ""
Write-Host "=== Feature flags premium por cuenta ==="
$featuresOk = $true
foreach ($line in $accounts) {
$parts = $line.Substring(8) -split '\|', 3
$state = if ($parts.Count -ge 3) { $parts[2] } else { "?" }
if ($state -ne "OK") { $featuresOk = $false }
Write-Host (" cuenta {0} ({1}): {2}" -f $parts[0], $parts[1], $(if ($state -eq "OK") { "todos activos" } else { "APAGADOS -> $state" }))
}
}
# --- 5) Veredicto ---------------------------------------------------------
Write-Host ""
if ($planOk -and ($featuresOk -eq $true)) {
Write-Host "[OK] Enterprise activo: plan=enterprise y feature flags premium completos."
}
elseif ($planOk -and ($null -eq $featuresOk)) {
Write-Host "[OK] plan=enterprise. Corre con -Deep para confirmar los feature flags por cuenta."
}
elseif ($planOk) {
Write-Host "[WARN] plan=enterprise pero hay feature flags premium apagados."
Write-Host " Corre: .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures"
}
else {
Write-Host "[FAIL] Enterprise NO activo (plan=$plan, quantity=$quantity)."
Write-Host " Corre: .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures"
Write-Host " Detalle del caso: docs/runbooks/chatwoot-update.md"
}
+9 -1
View File
@@ -96,12 +96,20 @@ if ($VerifyOnly) {
echo '--- onboot / startup on LXC $lxc ---' echo '--- onboot / startup on LXC $lxc ---'
grep -Ei 'onboot|startup' /etc/pve/lxc/$lxc.conf 2>/dev/null || echo '(onboot not set -> defaults to 0, will NOT auto-start)' grep -Ei 'onboot|startup' /etc/pve/lxc/$lxc.conf 2>/dev/null || echo '(onboot not set -> defaults to 0, will NOT auto-start)'
echo '--- guardian unit ---' echo '--- guardian unit ---'
systemctl is-enabled $svcName 2>/dev/null || echo '($svcName not installed/enabled)' printf 'enabled: '; systemctl is-enabled $svcName 2>/dev/null || echo '(not installed/enabled)'
printf 'state: '; systemctl is-active $svcName 2>/dev/null || true
printf 'result: '; systemctl show $svcName -p Result --value 2>/dev/null || true
printf 'timeout: '; systemctl show $svcName -p TimeoutStartUSec --value 2>/dev/null || true
echo '--- last boot outcome (journal) ---'
journalctl -u $svcName --no-pager -b 2>/dev/null | tail -n 6 || echo '(no journal for this boot)'
echo '--- guardian script present? ---' echo '--- guardian script present? ---'
test -x $binPath && echo 'present ($binPath)' || echo 'missing ($binPath)' test -x $binPath && echo 'present ($binPath)' || echo 'missing ($binPath)'
echo '--- last log lines ---' echo '--- last log lines ---'
tail -n 15 $logPath 2>/dev/null || echo '(no log yet)' tail -n 15 $logPath 2>/dev/null || echo '(no log yet)'
"@ "@
Write-Host ""
Write-Host "Healthy = enabled:enabled / state:active / result:success and a log run ending in 'done'." -ForegroundColor DarkGray
Write-Host "state:failed or a log run that stops after 'LXC ... running' means the guardian died mid-wait." -ForegroundColor DarkGray
return return
} }
+15
View File
@@ -0,0 +1,15 @@
param(
[Parameter(Mandatory = $true)]
[string]$BashCommand
)
$ErrorActionPreference = "Stop"
. "$PSScriptRoot\ProxmoxAgent.ps1"
$text = $BashCommand -replace "`r`n", "`n"
$bytes = [System.Text.Encoding]::UTF8.GetBytes($text)
$b64 = [Convert]::ToBase64String($bytes)
# On Proxmox host, run pct exec with base64 decoded directly inside container
$remoteCmd = "pct exec 102 -- bash -c 'echo " + $b64 + " | base64 -d | bash'"
Invoke-ProxmoxSshCommand -Command $remoteCmd
+1 -1
View File
@@ -10,7 +10,7 @@ function Get-ProxmoxConfig {
HostName = if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" } HostName = if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" }
Node = if ($env:PROXMOX_NODE) { $env:PROXMOX_NODE } else { "thinkcentre" } Node = if ($env:PROXMOX_NODE) { $env:PROXMOX_NODE } else { "thinkcentre" }
User = if ($env:PROXMOX_USER) { $env:PROXMOX_USER } else { "root" } User = if ($env:PROXMOX_USER) { $env:PROXMOX_USER } else { "root" }
SshKey = if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" } SshKey = if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { Join-Path $PSScriptRoot "..\keys\proxmox_ed25519" }
ApiBaseUrl = if ($env:PROXMOX_API_BASE_URL) { $env:PROXMOX_API_BASE_URL } else { "https://192.168.0.200:8006/api2/json" } ApiBaseUrl = if ($env:PROXMOX_API_BASE_URL) { $env:PROXMOX_API_BASE_URL } else { "https://192.168.0.200:8006/api2/json" }
ApiTokenHeader = $tokenHeader ApiTokenHeader = $tokenHeader
CoolifyLxc = if ($env:PROXMOX_COOLIFY_LXC) { $env:PROXMOX_COOLIFY_LXC } else { "102" } CoolifyLxc = if ($env:PROXMOX_COOLIFY_LXC) { $env:PROXMOX_COOLIFY_LXC } else { "102" }
+1 -1
View File
@@ -3,7 +3,7 @@
$env:PROXMOX_HOST = "192.168.0.200" $env:PROXMOX_HOST = "192.168.0.200"
$env:PROXMOX_NODE = "thinkcentre" $env:PROXMOX_NODE = "thinkcentre"
$env:PROXMOX_USER = "root" $env:PROXMOX_USER = "root"
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" $env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519" # relativo a la raíz del repo (≡ ~\.ssh\coolify_key)
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json" $env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw" $env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET" $env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
+102
View File
@@ -0,0 +1,102 @@
# ===========================================================================
# Deploy-OhDaddy.ps1
# Redeploy de oh-daddy (https://github.com/KenKaiii/oh-daddy) en Coolify
# (LXC 102) SIN depender del pull de Coolify (la imagen es local:
# oh-daddy-app:local, construida en el server - patron Deploy-SoloLeveling).
#
# Stack: app (Next.js 16) + db (postgres:17) + inngest (self-hosted v1.44.0)
# + inngest-db + inngest-redis. Servicio Coolify rzittzudkunwx8gilonn7tqe,
# proyecto "AI AGENCY" / production. Dominio: ohdaddy.urieljareth.org.
#
# Pre-requisitos:
# - .env.local.ps1 con COOLIFY_*, PROXMOX_* (los secretos OH_DADDY_* solo
# se necesitan si vas a re-aplicar envs; el deploy los lee del .env del
# servicio en disco).
# - stacks/oh-daddy/{Dockerfile,.dockerignore} en este repo (se inyectan
# al clone via base64).
#
# Uso:
# . .\.env.local.ps1
# .\scripts\apps\Deploy-OhDaddy.ps1 # build + up + schema + registro
# .\scripts\apps\Deploy-OhDaddy.ps1 -NoBuild # solo up + verificaciones
# ===========================================================================
[CmdletBinding()]
param(
[string]$ServiceUuid = 'rzittzudkunwx8gilonn7tqe',
[string]$Fqdn = 'https://ohdaddy.urieljareth.org',
[string]$Repo = 'https://github.com/KenKaiii/oh-daddy.git',
[string]$Image = 'oh-daddy-app:local',
[switch]$NoBuild
)
Set-Location 'H:\MegaSync\Proyectos\Proxmox & Coolify Manager'
. .\.env.local.ps1
$ErrorActionPreference = 'Stop'
function Invoke-OnServer([string]$scriptBody, [int]$TimeoutSec = 600) {
$body = ($scriptBody -replace "`r`n","`n") -replace "`r","`n"
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($body))
$cmd = "pct exec 102 -- sh -c 'echo $b64 | base64 -d | sh'"
& .\scripts\Invoke-ProxmoxSsh.ps1 -Command $cmd
}
$stackDir = 'H:\MegaSync\Proyectos\Proxmox & Coolify Manager\stacks\oh-daddy'
# ---------------------------------------------------------------- 1. BUILD ---
if (-not $NoBuild) {
Write-Host "==> Construyendo imagen en el server (clone + docker build)..." -ForegroundColor Cyan
$dfB64 = [Convert]::ToBase64String([IO.File]::ReadAllBytes((Join-Path $stackDir 'Dockerfile')))
$diB64 = [Convert]::ToBase64String([IO.File]::ReadAllBytes((Join-Path $stackDir '.dockerignore')))
$build = @"
set -e
BUILD=/tmp/oh-daddy-build
rm -rf "`$BUILD" /tmp/oh-daddy-build.log
mkdir -p "`$BUILD"; cd "`$BUILD"
git clone --depth 1 $Repo repo
cd repo
echo "HEAD: `$(git log -1 --oneline)"
echo '$dfB64' | base64 -d > Dockerfile
echo '$diB64' | base64 -d > .dockerignore
nohup sh -c 'cd /tmp/oh-daddy-build/repo && docker build --build-arg NEXT_PUBLIC_APP_URL=$Fqdn -t $Image . ; echo BUILD_EXIT=`$?' > /tmp/oh-daddy-build.log 2>&1 &
echo BUILD_STARTED
"@
Invoke-OnServer $build
Write-Host " (build en background; log: pct exec 102 -- tail -f /tmp/oh-daddy-build.log)" -ForegroundColor DarkGray
Write-Host "==> Esperando build (poll cada 30s, max 15min)..." -ForegroundColor Cyan
$deadline = (Get-Date).AddMinutes(15)
while ((Get-Date) -lt $deadline) {
Start-Sleep -Seconds 30
$st = Invoke-OnServer "set +e`ntail -3 /tmp/oh-daddy-build.log`ndocker images $Image --format '{{.ID}}'"
if ($st -match 'BUILD_EXIT=0') { Write-Host "OK: imagen $Image construida." -ForegroundColor Green; break }
if ($st -match 'BUILD_EXIT=([1-9][0-9]*)') { throw "Build fallo (exit $($Matches[1])). Log: /tmp/oh-daddy-build.log" }
Write-Host " ... compilando" -ForegroundColor DarkGray
}
if (-not $st) { throw "No se pudo confirmar el build." }
}
# -------------------------------------------------- 2. UP + PROXY + SCHEMA ---
Write-Host "==> Levantando stack, conectando proxy y aplicando schema..." -ForegroundColor Cyan
$up = @"
set -e
SVC=/data/coolify/services/$ServiceUuid
cd "`$SVC"
docker compose up -d
docker network connect $ServiceUuid coolify-proxy 2>&1 && echo PROXY_CONNECTED || echo "(proxy ya conectado)"
echo "=== SCHEMA (idempotente) ==="
docker exec -i db-$ServiceUuid psql -U ohdaddy -d ohdaddy < /tmp/oh-daddy-build/repo/db/schema.sql 2>&1 | grep -cE 'ERROR' | sed 's/^/errores_sql=/' || true
echo "=== PS ==="
sleep 15
docker compose ps --format '{{.Name}} | {{.Status}}'
"@
Invoke-OnServer $up
# ------------------------------------------- 3. REGISTRO INNGEST + SALUD ---
Write-Host "==> Re-registrando funciones en Inngest (PUT publico)..." -ForegroundColor Cyan
$code = & curl.exe -s -o NUL -w "%{http_code}" --max-time 60 -X PUT "$Fqdn/api/inngest"
Write-Host ("INNGEST_REGISTER=" + $code)
if ($code -ne '200') { Write-Host "AVISO: registro no confirmado. Reintentar cuando la app este healthy: curl -X PUT $Fqdn/api/inngest" -ForegroundColor Yellow }
Write-Host "=== HEALTH publico: $Fqdn ===" -ForegroundColor Cyan
$code2 = & curl.exe -s -o NUL -w "%{http_code}" --max-time 30 "$Fqdn/login"
Write-Host ("PUBLIC_LOGIN=" + $code2)
if ($code2 -eq '200') { Write-Host "OK: oh-daddy activo en $Fqdn" -ForegroundColor Green }
else { Write-Host "AVISO: /login no responde 200 ($code2). Logs: docker logs app-$ServiceUuid" -ForegroundColor Yellow }
+140
View File
@@ -0,0 +1,140 @@
#!/bin/bash
# Guard: mantiene la edicion enterprise de Chatwoot en LXC 102.
#
# Por que existe: Internal::CheckNewVersionsJob hace ping diario a
# hub.2.chatwoot.com (a las 16:16 UTC para este installation_identifier) y
# reescribe INSTALLATION_PRICING_PLAN con lo que responda el hub ('community'),
# y ademas Internal::ReconcilePlanConfigService apaga los 9 feature flags
# premium en TODAS las cuentas.
#
# No se bloquea el hub a proposito: ese mismo host relaya las notificaciones
# push del movil (ChatwootHub.send_push) porque FIREBASE_* esta vacio. Bloquearlo
# romperia el push. En vez de eso, este guard detecta el revert y lo deshace.
#
# Cheap por diseno: en el caso normal hace 1 SELECT y sale. Solo cuando detecta
# plan != enterprise levanta Rails para limpiar cache y reactivar flags.
#
# Instalado por el runbook docs/runbooks/chatwoot-update.md
# Cron: */15 * * * *
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
LXC=102
UUID=c11xzy2tx2cdapm32f5b89vy
DB_CT="postgres-$UUID"
APP_CT="chatwoot-$UUID"
LOG=/var/log/chatwoot-enterprise-guard.log
LOCK=/var/lock/chatwoot-enterprise-guard.lock
MAX_LOG=2097152
log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*" >> "$LOG"; }
# Rotacion simple para que el log no crezca sin control.
if [ -f "$LOG" ] && [ "$(stat -c %s "$LOG" 2>/dev/null || echo 0)" -gt "$MAX_LOG" ]; then
mv -f "$LOG" "$LOG.1"
fi
# Una sola instancia a la vez (el camino de reparacion tarda ~1 min).
exec 9>"$LOCK" || exit 0
flock -n 9 || exit 0
# --- 1) Chequeo barato: en que plan estamos ---------------------------------
# Se traen todas las filas y se filtra con grep a proposito: un WHERE con
# literales necesitaria comillas simples anidadas dentro de bash -lc '...' y eso
# es exactamente la clase de escaping que rompe estos scripts.
PLAN=$(pct exec "$LXC" -- docker exec -i "$DB_CT" bash -lc \
'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -F "|" -c "SELECT name, serialized_value::text FROM public.installation_configs"' \
2>/dev/null | grep '^INSTALLATION_PRICING_PLAN|')
if [ -z "$PLAN" ]; then
log "ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)"
exit 1
fi
case "$PLAN" in
*enterprise*)
# Caso normal: nada que hacer, y no ensuciamos el log.
exit 0
;;
esac
log "DETECTADO revert -> plan actual: $PLAN . Reparando..."
# --- 2) Reponer las 3 filas del parche --------------------------------------
cat > /tmp/cw-guard.sql <<'SQLEOF'
UPDATE public.installation_configs
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: enterprise\n"'
WHERE name = 'INSTALLATION_PRICING_PLAN';
UPDATE public.installation_configs
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: 10000\n"'
WHERE name = 'INSTALLATION_PRICING_PLAN_QUANTITY';
UPDATE public.installation_configs
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: e04t63ee-5gg8-4b94-8914-ed8137a7d938\n"'
WHERE name = 'INSTALLATION_IDENTIFIER';
SQLEOF
SQL_OUT=$(pct exec "$LXC" -- docker exec -i "$DB_CT" bash -lc \
'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -v ON_ERROR_STOP=1' \
< /tmp/cw-guard.sql 2>&1)
SQL_RC=$?
rm -f /tmp/cw-guard.sql
UPDATES=$(printf '%s\n' "$SQL_OUT" | grep -c '^UPDATE 1$')
if [ "$SQL_RC" -ne 0 ] || [ "$UPDATES" -ne 3 ]; then
log "ERROR: los UPDATE fallaron (rc=$SQL_RC, 'UPDATE 1'=$UPDATES). Salida: $SQL_OUT"
exit 1
fi
log "OK: 3/3 UPDATE aplicados"
# --- 3) Limpiar cache de GlobalConfig y reactivar flags premium -------------
# El UPDATE por SQL no dispara el after_commit :clear_cache de InstallationConfig,
# y GlobalConfig cachea en Redis con TTL de 1 dia: sin este paso la app puede
# seguir sirviendo 'community'. Ademas hay que reactivar los flags por cuenta,
# que viven en accounts.feature_flags (bitmask) y el SQL de arriba no toca.
cat > /tmp/cw-guard.rb <<'RBEOF'
PREMIUM = %w[
disable_branding audit_logs sla custom_roles
captain_integration captain_integration_v2 captain_document_auto_sync
csat_review_notes conversation_required_attributes
]
GlobalConfig.clear_cache
Account.find_each do |account|
account.enable_features!(*PREMIUM)
account.reload
pend = PREMIUM.reject { |f| account.feature_enabled?(f) }
puts "account=#{account.id} pendientes=#{pend.empty? ? 'ninguno' : pend.join(',')}"
end
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
RBEOF
pct push "$LXC" /tmp/cw-guard.rb /tmp/cw-guard.rb
pct exec "$LXC" -- docker cp /tmp/cw-guard.rb "$APP_CT":/tmp/cw-guard.rb
RB_OUT=$(pct exec "$LXC" -- docker exec -i "$APP_CT" bundle exec rails runner /tmp/cw-guard.rb 2>&1)
RB_RC=$?
pct exec "$LXC" -- docker exec -i "$APP_CT" rm -f /tmp/cw-guard.rb
pct exec "$LXC" -- rm -f /tmp/cw-guard.rb
rm -f /tmp/cw-guard.rb
RESULT=$(printf '%s\n' "$RB_OUT" | grep -E '^(account=|self_hosted_enterprise=)' | tr '\n' ' ')
if [ "$RB_RC" -ne 0 ]; then
log "ERROR: rails runner fallo (rc=$RB_RC). Salida: $RB_OUT"
exit 1
fi
case "$RESULT" in
*"self_hosted_enterprise=true"*)
case "$RESULT" in
*"pendientes=ninguno"*)
log "REPARADO: $RESULT"
exit 0
;;
esac
log "PARCIAL: enterprise activo pero quedaron flags pendientes -> $RESULT"
exit 1
;;
esac
log "ERROR: tras reparar, self_hosted_enterprise no dio true -> $RESULT"
exit 1
+9 -1
View File
@@ -6,12 +6,20 @@ Description=Auto-start Coolify LXC + Docker stack + Cloudflare tunnel after boot
# policies, not a replacement for them. # policies, not a replacement for them.
After=pve-guests.service network-online.target After=pve-guests.service network-online.target
Wants=pve-guests.service network-online.target Wants=pve-guests.service network-online.target
# Bound retries so a genuinely broken stack doesn't loop forever.
StartLimitIntervalSec=3600
StartLimitBurst=3
[Service] [Service]
Type=oneshot Type=oneshot
ExecStart=/usr/local/bin/coolify-autostart.sh ExecStart=/usr/local/bin/coolify-autostart.sh
RemainAfterExit=yes RemainAfterExit=yes
TimeoutStartSec=300 # Must stay above the script's own budget (COOLIFY_MAX_WAIT 600 + COOLIFY_SETTLE
# 180 + step overhead). The old 300 s killed the guardian mid-wait on every real
# boot: the `coolify` container only starts ~7m40s after power-on.
TimeoutStartSec=1200
Restart=on-failure
RestartSec=60
[Install] [Install]
WantedBy=multi-user.target WantedBy=multi-user.target
+61 -16
View File
@@ -8,20 +8,40 @@
# #
# Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and # Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and
# invoked by the systemd unit coolify-autostart.service on multi-user.target. # invoked by the systemd unit coolify-autostart.service on multi-user.target.
#
# Timing note (measured on the 2026-08-07 boots): pve-guests takes ~78 s to
# start CT 102, the Docker daemon inside it only answers ~4-5 min after boot,
# and the last core container (`coolify`) starts ~7m40s after boot. Every wait
# here is therefore wall-clock based and generously sized; the systemd unit's
# TimeoutStartSec must stay above MAX_WAIT + SETTLE.
# ----------------------------------------------------------------------------- # -----------------------------------------------------------------------------
set -uo pipefail set -uo pipefail
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
LXC_ID="${COOLIFY_LXC:-102}" LXC_ID="${COOLIFY_LXC:-102}"
LOG="/var/log/coolify-autostart.log" LOG="${COOLIFY_LOG:-/var/log/coolify-autostart.log}" # override to dry-run without touching the real log
MAX_WAIT="${COOLIFY_MAX_WAIT:-180}" # seconds to wait for docker inside the LXC MAX_WAIT="${COOLIFY_MAX_WAIT:-600}" # wall-clock seconds to wait for docker
SETTLE="${COOLIFY_SETTLE:-180}" # grace for docker to start its own containers
PROBE_TIMEOUT="${COOLIFY_PROBE_TIMEOUT:-20}" # hard timeout for every call into the LXC
# Dependencies first, then the app, proxy and tunnel. These all normally come # Dependencies first, then the app, proxy and tunnel. These all normally come
# up on their own via Docker restart policies; this loop only heals the ones # up on their own via Docker restart policies; this loop only heals the ones
# that didn't (e.g. restart=no, or a wedged start). # that didn't (e.g. restart=no, or a wedged start).
CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared) CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared)
failures=0
log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; } log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; }
now() { date +%s; }
# Every call into the LXC gets a hard timeout. During a cold boot the Docker
# daemon is busy starting ~50 containers and a single blocking `docker` call
# used to eat the entire budget silently, so the guardian was killed by systemd
# before it ever reached the container checks.
lxc() { timeout "$PROBE_TIMEOUT" pct exec "$LXC_ID" -- "$@"; }
# true | false | unknown (unknown = inspect failed, container may not exist yet)
container_state() { lxc docker inspect -f '{{.State.Running}}' "$1" 2>/dev/null || echo unknown; }
log "=== coolify-autostart start (LXC ${LXC_ID}) ===" log "=== coolify-autostart start (LXC ${LXC_ID}) ==="
@@ -33,35 +53,48 @@ if [ "$status" != "running" ]; then
log "pct start issued" log "pct start issued"
else else
log "ERROR: pct start ${LXC_ID} failed" log "ERROR: pct start ${LXC_ID} failed"
failures=$((failures + 1))
fi fi
else else
log "LXC ${LXC_ID} already running" log "LXC ${LXC_ID} already running"
fi fi
# 2. Wait for the Docker daemon inside the LXC to respond ---------------------- # 2. Wait for the Docker daemon inside the LXC to respond ----------------------
waited=0 # Wall-clock deadline (not a sleep counter) and a cheap probe: `docker
until pct exec "$LXC_ID" -- docker info >/dev/null 2>&1; do # version` hits /version, while `docker info` enumerates every container and
if [ "$waited" -ge "$MAX_WAIT" ]; then # plugin and stalls for minutes on a loaded daemon.
log "ERROR: docker not ready after ${MAX_WAIT}s -> aborting" start_ts="$(now)"
deadline=$((start_ts + MAX_WAIT))
until lxc docker version --format '{{.Server.Version}}' >/dev/null 2>&1; do
if [ "$(now)" -ge "$deadline" ]; then
log "ERROR: docker not ready after ${MAX_WAIT}s (wall clock) -> aborting"
exit 1 exit 1
fi fi
sleep 5 sleep 5
waited=$((waited + 5))
done done
log "docker ready after ${waited}s" log "docker ready after $(( $(now) - start_ts ))s"
# 3. Ensure the core Coolify containers + tunnel container are running --------- # 3. Ensure the core Coolify containers + tunnel container are running ---------
# Docker's own restart policies bring these up over several minutes, so poll
# each one until SETTLE expires before forcing a start.
settle_deadline=$(($(now) + SETTLE))
for c in "${CORE_CONTAINERS[@]}"; do for c in "${CORE_CONTAINERS[@]}"; do
st="$(pct exec "$LXC_ID" -- docker inspect -f '{{.State.Running}}' "$c" 2>/dev/null || echo missing)" st="$(container_state "$c")"
while [ "$st" != "true" ] && [ "$(now)" -lt "$settle_deadline" ]; do
sleep 5
st="$(container_state "$c")"
done
case "$st" in case "$st" in
true) log "container ${c}: running" ;; true) log "container ${c}: running" ;;
missing) log "WARN container ${c}: not found (skipping)" ;;
*) *)
log "container ${c}: running=${st} -> starting" log "container ${c}: running=${st} -> starting"
if pct exec "$LXC_ID" -- docker start "$c" >/dev/null 2>&1; then if lxc docker start "$c" >/dev/null 2>&1; then
log "started ${c}" log "started ${c}"
elif [ "$st" = "unknown" ]; then
log "WARN container ${c}: not found (skipping)"
else else
log "ERROR: could not start ${c}" log "ERROR: could not start ${c}"
failures=$((failures + 1))
fi fi
;; ;;
esac esac
@@ -70,12 +103,24 @@ done
# 4. Ensure the systemd cloudflared tunnel inside the LXC is up ---------------- # 4. Ensure the systemd cloudflared tunnel inside the LXC is up ----------------
# (second connector to the same tunnel; belt-and-suspenders alongside the # (second connector to the same tunnel; belt-and-suspenders alongside the
# cloudflared Docker container above.) # cloudflared Docker container above.)
if pct exec "$LXC_ID" -- systemctl is-enabled cloudflared >/dev/null 2>&1; then if lxc systemctl is-enabled cloudflared >/dev/null 2>&1; then
if pct exec "$LXC_ID" -- systemctl start cloudflared >/dev/null 2>&1; then if ! lxc systemctl start cloudflared >/dev/null 2>&1; then
log "cloudflared.service ensured up"
else
log "WARN: cloudflared.service start returned non-zero" log "WARN: cloudflared.service start returned non-zero"
fi fi
log "cloudflared.service is-active=$(lxc systemctl is-active cloudflared 2>/dev/null || echo unknown)"
else
log "WARN: cloudflared.service not enabled inside LXC ${LXC_ID}"
fi fi
log "=== coolify-autostart done ===" # 5. Prove the tunnel actually reached Cloudflare this boot --------------------
# Without this the log said "ensured up" even when no connector registered.
conns="$(lxc journalctl -u cloudflared -b --no-pager 2>/dev/null | grep -c 'Registered tunnel connection' || true)"
conns="${conns//[!0-9]/}" # grep -c exits 1 on zero matches; keep only the digits
conns="${conns:-0}"
log "cloudflared.service registered tunnel connections this boot: ${conns}"
if [ "$conns" -eq 0 ]; then
log "WARN: no tunnel connection registered yet (edge may still be connecting)"
fi
log "=== coolify-autostart done (failures=${failures}) ==="
[ "$failures" -eq 0 ] || exit 1
@@ -0,0 +1,131 @@
# Evolution Go (API WhatsApp en Go) — stack para el host casero.
#
# Fuente: https://github.com/evolution-foundation/evolution-go (clon local en
# projects/evolution-go, tag 0.7.2). Imagen publicada: evoapicloud/evolution-go.
#
# Por qué este compose y no el de upstream (docker/examples/docker-compose.yml):
# - publica `ports:` 4000 y 5432 al host; aquí el 80/443 y el resto de puertos
# los gestiona Traefik/Coolify y no hay que publicar nada (§2.3 del contrato).
# - monta ./init-db.sql por ruta local: imposible en Coolify, el compose se
# guarda como docker_compose_raw en la DB, no hay árbol de archivos. No hace
# falta: ensureDBExists() (pkg/config/config.go) crea la DB del DSN al arrancar.
# - usa vars que el código de 0.7.2 ya no lee (WADEBUG/LOGTYPE); las reales son
# DEBUG_ENABLED/LOG_TYPE (pkg/config/env/env.go).
#
# Detalles verificados contra el código 0.7.2:
# - GateMiddleware (pkg/core/c0.go) devuelve 503 en TODO hasta activar licencia,
# EXCEPTO /server/ok, /manager, /assets, /license/*, /swagger, /ws. El
# healthcheck usa /server/ok (200 siempre) para no dejar al contenedor sin
# ruta en Traefik antes de activar la licencia desde el Manager.
# - whatsmeow guarda las sesiones SQLite en /app/dbdata (exPath = /app):
# SIEMPRE volumen con nombre o cada redeploy desvincula los teléfonos.
# - POSTGRES_AUTH_DB (URI completa) es OBLIGATORIA aunque parezca opcional:
# con string vacío initPostgresAuthDB() devuelve (nil, nil) —sin error— y
# NewPollService() hace panic por nil deref sobre el *sql.DB (bug upstream
# 0.7.2, cmd/evolution-go/main.go:300 + pkg/poll/service/poll_service.go:39).
# La URI interpola ${POSTGRES_PASSWORD}: Coolify sustituye al deployear,
# el secreto vive solo en la env del servicio.
# - POSTGRES_USERS_DB en URI para seguir el contrato upstream (con las vars
# discretas también funciona; la DB evogo_users se auto-crea igual).
# - DATABASE_SAVE_MESSAGES y GLOBAL_API_KEY son obligatorias (panicIfEmpty).
# - SERVER_PORT no tiene default en el código: fijarlo siempre.
#
# Contrato: docs/AGENTS-coolify-apps.md
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
# - DB hermana por nombre de servicio, nunca localhost (§2.1)
# - secretos (${VAR} sin default) llegan como env de Coolify, jamás aquí (§2.5)
# - healthchecks con start_period holgado: el primer arranque aquí tarda
# minutos (docs/casos/coolify-servicio-nuevo-503-no-available-server.md)
services:
evolution-go:
image: 'evoapicloud/evolution-go:0.7.2'
environment:
SERVER_PORT: '8080'
CLIENT_NAME: '${CLIENT_NAME:-evolution}'
# Obligatoria (panicIfEmpty). Secreto: inyectada por Coolify.
GLOBAL_API_KEY: '${GLOBAL_API_KEY}'
# Obligatoria y no vacía (panicIfEmpty).
DATABASE_SAVE_MESSAGES: '${DATABASE_SAVE_MESSAGES:-false}'
# OBLIGATORIA en URI (ver cabecera: vacía = panic nil deref en 0.7.2).
# Coolify interpola ${POSTGRES_PASSWORD} del env del servicio al deploy.
POSTGRES_AUTH_DB: 'postgresql://${POSTGRES_USER:-evolution}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_auth?sslmode=disable'
POSTGRES_USERS_DB: 'postgresql://${POSTGRES_USER:-evolution}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_users?sslmode=disable'
POSTGRES_HOST: evolution-postgres
POSTGRES_PORT: '5432'
POSTGRES_USER: '${POSTGRES_USER:-evolution}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD}'
POSTGRES_DB: '${POSTGRES_DB:-evogo_users}'
# Nombres reales en 0.7.2 (env.go): DEBUG_ENABLED / LOG_TYPE.
DEBUG_ENABLED: '${DEBUG_ENABLED:-INFO}'
LOG_TYPE: '${LOG_TYPE:-console}'
WEBHOOK_FILES: 'true'
CONNECT_ON_STARTUP: 'false'
OS_NAME: 'Linux'
EVENT_IGNORE_GROUP: 'false'
EVENT_IGNORE_STATUS: 'true'
QRCODE_MAX_COUNT: '5'
# AMQP/NATS/MinIO/WEBHOOK_URL apagados a propósito: nada de colas ni
# objeto-storage extra en este host; el Manager funciona igual.
AMQP_GLOBAL_ENABLED: 'false'
NATS_GLOBAL_ENABLED: 'false'
MINIO_ENABLED: 'false'
# Sin bloque `ports:` — Traefik llega al 8080 interno (§2.3). FQDN sin
# puerto + único `expose`: el mismo patrón verificado del stack firecrawl.
expose:
- '8080'
volumes:
# Sesiones whatsmeow (SQLite): perder esto = re-escanear QR de todos los
# teléfonos en cada redeploy (§5).
- 'evolution-data:/app/dbdata'
- 'evolution-logs:/app/logs'
depends_on:
evolution-postgres:
condition: service_healthy
# Imagen alpine: wget es el de busybox (curl no viene instalado en 0.7.2).
# /server/ok responde 200 sin licencia activa — ver comentario de cabecera.
healthcheck:
test: ['CMD', 'wget', '-q', '-O', '/dev/null', 'http://127.0.0.1:8080/server/ok']
interval: 15s
timeout: 10s
retries: 20
start_period: 300s
mem_limit: 1g
memswap_limit: 1g
cpus: 1.5
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
evolution-postgres:
image: 'postgres:16-alpine'
environment:
POSTGRES_USER: '${POSTGRES_USER:-evolution}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD}'
POSTGRES_DB: '${POSTGRES_DB:-evogo_users}'
expose:
- '5432'
healthcheck:
test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-evolution} -d ${POSTGRES_DB:-evogo_users}']
interval: 15s
timeout: 10s
retries: 20
start_period: 180s
volumes:
- 'evolution-postgres:/var/lib/postgresql/data'
mem_limit: 1g
memswap_limit: 1g
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 5m
max-file: '2'
volumes:
evolution-data: {}
evolution-logs: {}
evolution-postgres: {}
+189
View File
@@ -0,0 +1,189 @@
# Firecrawl - stack mínimo con imágenes precompiladas, para el host casero.
#
# Por qué no se usa el compose de upstream (firecrawl/firecrawl, rama main):
# - construye 3 servicios desde fuente (apps/api, apps/playwright-service-ts,
# apps/nuq-postgres). Compilar Chromium en un disco a ~26 ms/escritura es el
# peor caso posible en este host.
# - añade FoundationDB (+ un init) que solo se usan si NUQ_BACKEND está puesto.
# - pide mem_limit 8G en api y 4G en playwright. El LXC tiene 4 cores.
#
# Aquí todo son imágenes ya publicadas: cero builds. FoundationDB queda fuera y
# NUQ_BACKEND se deja vacío, que es su modo por defecto.
#
# Contrato: docs/AGENTS-coolify-apps.md
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
# - hermanos por nombre de servicio, nunca localhost (§2.1)
# - volumen con nombre para lo que debe sobrevivir a un redeploy (§5)
# - healthchecks con start_period holgado: el primer arranque aquí tarda
# minutos (ver docs/casos/coolify-servicio-nuevo-503-no-available-server.md)
# - sin secretos en el archivo: llegan como variables de entorno (§2.5)
services:
api:
image: 'ghcr.io/firecrawl/firecrawl:2.10.19'
environment:
HOST: 0.0.0.0
PORT: '3002'
INTERNAL_PORT: '3002'
WORKER_PORT: '3005'
EXTRACT_WORKER_PORT: '3004'
ENV: local
# Hermanos por nombre de servicio (§2.1)
REDIS_URL: 'redis://redis:6379'
REDIS_RATE_LIMIT_URL: 'redis://redis:6379'
PLAYWRIGHT_MICROSERVICE_URL: 'http://playwright-service:3000/scrape'
NUQ_RABBITMQ_URL: 'amqp://rabbitmq:5672'
POSTGRES_HOST: nuq-postgres
POSTGRES_PORT: '5432'
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
USE_DB_AUTHENTICATION: 'false'
# Vacío a propósito: con NUQ_BACKEND sin definir, FoundationDB no se usa.
NUQ_BACKEND: ''
# Concurrencia recortada para 4 cores (upstream trae 8/10/5/5).
NUM_WORKERS_PER_QUEUE: '${NUM_WORKERS_PER_QUEUE:-2}'
CRAWL_CONCURRENT_REQUESTS: '${CRAWL_CONCURRENT_REQUESTS:-3}'
MAX_CONCURRENT_JOBS: '${MAX_CONCURRENT_JOBS:-2}'
BROWSER_POOL_SIZE: '${BROWSER_POOL_SIZE:-2}'
HARNESS_STARTUP_TIMEOUT_MS: '${HARNESS_STARTUP_TIMEOUT_MS:-180000}'
LOGGING_LEVEL: '${LOGGING_LEVEL:-info}'
# Sin esto el worker responde "Can't accept connection due to RAM/CPU
# load" y rechaza todo: el umbral por defecto (0.8) se supera constantemente
# en un host compartido como este.
MAX_RAM: '${MAX_RAM:-0.95}'
MAX_CPU: '${MAX_CPU:-0.95}'
# Secretos: inyectados por Coolify, nunca literales aquí (§2.5)
BULL_AUTH_KEY: '${BULL_AUTH_KEY}'
TEST_API_KEY: '${TEST_API_KEY}'
OPENAI_API_KEY: '${OPENAI_API_KEY}'
OPENAI_BASE_URL: '${OPENAI_BASE_URL}'
MODEL_NAME: '${MODEL_NAME}'
MODEL_EMBEDDING_NAME: '${MODEL_EMBEDDING_NAME}'
SEARXNG_ENDPOINT: '${SEARXNG_ENDPOINT}'
# Sin bloque `ports:` — Traefik llega al puerto interno (§2.3)
expose:
- '3002'
depends_on:
redis:
condition: service_started
playwright-service:
condition: service_started
rabbitmq:
condition: service_healthy
nuq-postgres:
condition: service_healthy
# Verificado dentro de la imagen: NO trae wget ni nc, solo curl. Y /test,
# /health y /v1/health dan 404; la raiz da 200. Un healthcheck con wget
# falla siempre y deja el contenedor sin ruta en Traefik -> 503.
healthcheck:
test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/']
interval: 15s
timeout: 10s
retries: 20
start_period: 300s
ulimits:
nofile:
soft: 65535
hard: 65535
extra_hosts:
- 'host.docker.internal:host-gateway'
mem_limit: 3g
memswap_limit: 3g
cpus: 2.0
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
playwright-service:
image: 'ghcr.io/firecrawl/playwright-service:latest'
environment:
PORT: '3000'
MAX_CONCURRENT_PAGES: '${CRAWL_CONCURRENT_REQUESTS:-3}'
BLOCK_MEDIA: '${BLOCK_MEDIA:-true}'
ALLOW_LOCAL_WEBHOOKS: '${ALLOW_LOCAL_WEBHOOKS:-false}'
expose:
- '3000'
# Chromium escribe mucho en /tmp; en tmpfs no toca el disco lento.
tmpfs:
- '/tmp/.cache:noexec,nosuid,size=512m'
mem_limit: 2g
memswap_limit: 2g
cpus: 1.5
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
redis:
image: 'redis:alpine'
command: 'redis-server --bind 0.0.0.0 --save "" --appendonly no'
expose:
- '6379'
healthcheck:
test: ['CMD', 'redis-cli', 'ping']
interval: 15s
timeout: 5s
retries: 10
start_period: 60s
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 5m
max-file: '2'
rabbitmq:
image: 'rabbitmq:3-management'
expose:
- '5672'
healthcheck:
test: ['CMD', 'rabbitmq-diagnostics', '-q', 'check_running']
interval: 15s
timeout: 15s
retries: 20
start_period: 180s
volumes:
- 'firecrawl-rabbitmq:/var/lib/rabbitmq'
mem_limit: 1g
memswap_limit: 1g
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 5m
max-file: '2'
nuq-postgres:
image: 'ghcr.io/firecrawl/nuq-postgres:latest'
environment:
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
expose:
- '5432'
healthcheck:
test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-postgres} -d ${POSTGRES_DB:-postgres}']
interval: 15s
timeout: 10s
retries: 20
start_period: 180s
volumes:
- 'firecrawl-postgres:/var/lib/postgresql/data'
mem_limit: 1g
memswap_limit: 1g
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
volumes:
firecrawl-postgres: {}
firecrawl-rabbitmq: {}
+6
View File
@@ -0,0 +1,6 @@
.git
node_modules
.next
*.md
.gg
tests
+34
View File
@@ -0,0 +1,34 @@
# oh-daddy (Next.js 16) - production image, built on the Coolify server.
# No upstream Dockerfile exists (repo targets Railway/nixpacks), so this
# replicates railway.json's startCommand contract: bash scripts/start.sh
# (backgrounds the Inngest post-deploy re-sync, then execs `npm run start`).
FROM node:22-alpine AS deps
WORKDIR /app
COPY package.json package-lock.json ./
RUN npm ci
FROM node:22-alpine AS proddeps
WORKDIR /app
COPY package.json package-lock.json ./
RUN npm ci --omit=dev
FROM node:22-alpine AS build
WORKDIR /app
ENV NEXT_TELEMETRY_DISABLED=1
COPY --from=deps /app/node_modules ./node_modules
COPY . .
ARG NEXT_PUBLIC_APP_URL
ENV NEXT_PUBLIC_APP_URL=$NEXT_PUBLIC_APP_URL
RUN npm run build
FROM node:22-alpine AS runner
WORKDIR /app
RUN apk add --no-cache bash
ENV NODE_ENV=production PORT=3000 NEXT_TELEMETRY_DISABLED=1
COPY --from=proddeps /app/node_modules ./node_modules
COPY --from=build /app/.next ./.next
COPY --from=build /app/public ./public
COPY --from=build /app/package.json ./package.json
COPY --from=build /app/scripts ./scripts
EXPOSE 3000
CMD ["bash", "scripts/start.sh"]
+108
View File
@@ -0,0 +1,108 @@
# oh-daddy stack for Coolify (service type: docker compose, prebuilt/local images)
# Upstream: https://github.com/KenKaiii/oh-daddy
# App image `oh-daddy-app:local` is built on the server (Deploy-OhDaddy flow,
# same pattern as Deploy-SoloLeveling.ps1) - Coolify's own deploy cannot pull it.
# All secrets come from Coolify service env vars (is_literal) / .env in the
# service dir; nothing is hardcoded here.
#
# Hard rules honored (docs/AGENTS-coolify-apps.md):
# - no published 80/443; Traefik routes via SERVICE_FQDN_APP_3000 -> port 3000
# - siblings reached by Docker service name (db, inngest, inngest-db, inngest-redis)
# - named volumes for everything that must survive redeploys
services:
app:
image: oh-daddy-app:local
restart: unless-stopped
environment:
- SERVICE_FQDN_APP_3000
- PORT=3000
- NODE_ENV=production
- DATABASE_URL=${DATABASE_URL}
- APP_ENCRYPTION_KEY=${APP_ENCRYPTION_KEY}
- ADMIN_PASSWORD=${ADMIN_PASSWORD}
- INNGEST_BASE_URL=${INNGEST_BASE_URL}
- INNGEST_SIGNING_KEY=${INNGEST_SIGNING_KEY}
- INNGEST_EVENT_KEY=${INNGEST_EVENT_KEY}
- NEXT_PUBLIC_APP_URL=${NEXT_PUBLIC_APP_URL}
expose:
- "3000"
healthcheck:
test: ["CMD", "wget", "-qO-", "http://127.0.0.1:3000/login"]
interval: 30s
timeout: 5s
retries: 3
start_period: 60s
depends_on:
db:
condition: service_healthy
db:
image: postgres:17-alpine
restart: unless-stopped
environment:
- POSTGRES_DB=ohdaddy
- POSTGRES_USER=ohdaddy
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD}
volumes:
- oh-daddy-db-data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U ohdaddy -d ohdaddy"]
interval: 5s
timeout: 5s
retries: 12
# Self-hosted Inngest engine (NOT Inngest Cloud) - internal only, no FQDN.
inngest:
image: inngest/inngest:v1.44.0
restart: unless-stopped
command: ["inngest", "start"]
environment:
- INNGEST_SIGNING_KEY=${INNGEST_SIGNING_KEY}
- INNGEST_EVENT_KEY=${INNGEST_EVENT_KEY}
- INNGEST_POSTGRES_URI=${INNGEST_POSTGRES_URI}
- INNGEST_REDIS_URI=redis://inngest-redis:6379
expose:
- "8288"
healthcheck:
test: ["CMD", "inngest", "alpha", "doctor", "healthcheck"]
interval: 30s
timeout: 10s
retries: 3
start_period: 40s
depends_on:
inngest-db:
condition: service_healthy
inngest-redis:
condition: service_healthy
inngest-db:
image: postgres:17-alpine
restart: unless-stopped
environment:
- POSTGRES_DB=inngest
- POSTGRES_USER=inngest
- POSTGRES_PASSWORD=${INNGEST_DB_PASSWORD}
volumes:
- oh-daddy-inngest-pg-data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U inngest -d inngest"]
interval: 5s
timeout: 5s
retries: 12
inngest-redis:
image: redis:7-alpine
restart: unless-stopped
volumes:
- oh-daddy-inngest-redis-data:/data
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 5s
timeout: 3s
retries: 5
volumes:
oh-daddy-db-data: {}
oh-daddy-inngest-pg-data: {}
oh-daddy-inngest-redis-data: {}
+22
View File
@@ -0,0 +1,22 @@
# OpenSEO — env vars para Coolify (template, sin secretos reales).
#
# Copia este archivo a `stacks/open-seo/.env.coolify`, rellena lo que aplique,
# y pásalo a `New-CoolifyService.ps1 -EnvFile`. El script empuja cada línea
# KEY=VALUE como env var del servicio; las referencias `${VAR}` dentro del
# compose se resuelven en runtime desde esas vars.
#
# Nunca commitees el archivo `.env.coolify` real — está en .gitignore.
# SEO data (opcional). Base64 de `email:password` de dataforseo.com — NO la
# dashboard API key. Vacío = la app arranca, los workflows SEO muestran "no
# data". Ver https://openseo.so/docs/DATAFORSEO_API_KEY.md
DATAFORSEO_API_KEY=
# Telemetry opt-out. "1" para desactivar el heartbeat anónimo y el beacon de
# fallo de preflight. Por defecto encendido.
OPENSEO_TELEMETRY_DISABLED=1
DO_NOT_TRACK=1
# SAM, el agente SEO dentro de la app (opcional). Oculto si vacío.
OPENROUTER_API_KEY=
OPENROUTER_MODEL=
+154
View File
@@ -0,0 +1,154 @@
# OpenSEO — stack self-host en Coolify
> Resuelto el **2026-08-27** contra el host real.
> Target: **LXC 102** (`coolify`), servicio compose, FQDN
> **`https://openseo.urieljareth.org`**.
> Imagen: **`ghcr.io/every-app/open-seo:sha-c469a48`** (mismo SHA que el deploy
> fallido anterior — esta vez la build de Vite sí corre, en el entrypoint).
---
## Por qué existe este stack
El deploy previo (`uuid kj0kccsb4d46tm0d6qe6docy`, `name=open-seo:main-...`,
deployment `fkojyfkqzp69hcba6miy8oer`) terminó con Coolify marcando verde y
**Caddy respondiendo 404 a todo**. Diagnóstico:
- `build_pack=railpack` clonó `every-app/open-seo@main`, construyó imagen local
con el mismo SHA `c469a48ae90ab58413b198fe3d1ac1aa90a9b070` y la cacheó.
- En el redeploy: `No build configuration changed & image found (...) Build
step skipped` → la imagen cacheada **no tenía `/app/dist`** (los artefactos
del build de Vite) y Coolify la reusó.
- Caddy (`/Caddyfile` con `root * /app/dist` + SPA fallback a `/index.html`)
no encontró nada y devolvió 404 a `/`, `/robots.txt`, `/health`.
- El README upstream lo dice textual: *"We recommend self-hosting with
Cloudflare as opposed to Railway, Coolify or Dokploy. We plan to make it
simpler to host on those platforms in the next few months."*
**Solución:** dejar de seguir upstream y consumir la **imagen prebuilt** que el
propio equipo publica en GHCR. Esa imagen tiene la cadena correcta:
`docker-entrypoint.sh` corre preflight → migrations → `pnpm run build` (que sí
genera `/app/dist`) → `vite preview` en el puerto `3001`. Y usa un fingerprint
para no reconstruir cuando los env vars relevantes no cambiaron.
Stack: **un solo servicio** (OpenSEO es self-contained: SQLite vía workerd en
`/app/.wrangler`, volumen `openseo-data`). Sin DB externa.
---
## Archivos
| Archivo | Para qué |
|---|---|
| `docker-compose.coolify.yml` | Compose que consume Coolify vía `POST /services` |
| `.env.example` | Template de env vars (sin secretos) |
| `.env.coolify` | **No committed.** Lo crea el operador con `cp .env.example .env.coolify` y rellena |
---
## Variables de entorno
Hardcoded en el compose (porque son decisión de arquitectura, no secretos):
| Var | Valor | Por qué |
|---|---|---|
| `PORT` | `3001` | Es donde escucha `vite preview` (per `Dockerfile.selfhost`) |
| `AUTH_MODE` | `local_noauth` | Single admin, sin pantalla de login. Aquí no tenemos `TEAM_DOMAIN`/`POLICY_AUD` de Cloudflare Access |
| `ALLOWED_HOST` | `openseo.urieljareth.org` | Sin esto, Vite bloquea toda petición externa con "Blocked request" |
| `CLOUDFLARE_INCLUDE_PROCESS_ENV` | `true` | Lo exige el runtime workerd para que process.env llegue a los bindings |
Suministradas vía `.env.coolify` (env vars del servicio en Coolify):
| Var | Default | Efecto |
|---|---|---|
| `DATAFORSEO_API_KEY` | vacío | **WARN** del preflight (no FAIL). Vacío = la app arranca, los workflows SEO devuelven "no data" |
| `OPENSEO_TELEMETRY_DISABLED` | `1` | Apaga el heartbeat anónimo |
| `DO_NOT_TRACK` | `1` | Alias del anterior |
| `OPENROUTER_API_KEY` | vacío | Habilita a SAM (el agente SEO integrado) si se setea |
| `OPENROUTER_MODEL` | vacío | Modelo a usar con SAM |
---
## Deploy
### 1. (Manual, una sola vez) Ingress del túnel de Cloudflare
El token de Cloudflare **no está** en `.env.local.ps1`, así que esto se hace en
el dashboard:
1. Cloudflare Zero Trust → Networks → Tunnels → tunnel `urieljareth` →
Configure → Public hostname.
2. Add a public hostname:
- Subdomain: `openseo`
- Domain: `urieljareth.org`
- Service: **HTTP** (no HTTPS, lo gestiona Coolify/Traefik)
- URL: `coolify.urieljareth.org` (o la IP interna del proxy de Coolify —
misma que usan los demás subdominios)
### 2. Crear el servicio en Coolify
```powershell
. .\.env.local.ps1
# Crear el archivo de env real (gitignored)
Copy-Item .\stacks\open-seo\.env.example .\stacks\open-seo\.env.coolify
# Editar .\stacks\open-seo\.env.coolify si quieres setear DATAFORSEO_API_KEY
.\deploy_skill\scripts\New-CoolifyService.ps1 `
-AppPath .\stacks\open-seo `
-AppName open-seo `
-Fqdn https://openseo.urieljareth.org `
-PrimaryService app `
-ProjectName "AI AGENCY" -EnvironmentName production `
-EnvFile .\stacks\open-seo\.env.coolify `
-InstantDeploy
```
### 3. Esperar al primer arranque
El primer `up` tarda **1-2 min**: preflight + migrations + vite build + arranque
de `vite preview`. Traefik no enruta hasta que el contenedor esté `healthy`.
```powershell
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
```
### 4. Verificar
```powershell
# 1. Endpoint público responde (TLS emitido, Traefik enrutando)
curl.exe -k -sSI https://openseo.urieljareth.org/
# 2. Status del contenedor
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -Filter openseo
# 3. Logs del entrypoint (debería verse "Preflight passed")
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker logs <cont> --tail 60"
# 4. Preflight reporta lo que falta
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> wget -qO- http://127.0.0.1:3001/api/health"
```
---
## Rollback / limpieza
| Acción | Comando |
|---|---|
| Parar la app rota original | `docker stop kj0kccsb4d46tm0d6qe6docy-055629993145` (vía `pct exec 102 --`) |
| Borrar la app rota de Coolify | UI → Service `kj0kccsb4d46tm0d6qe6docy` → Delete |
| Re-deployar | UI → Service `open-seo` → Redeploy |
---
## Cosas que caducan
- **El tag `:sha-c469a48` se queda viejo.** Cuando el upstream publique un SHA
más reciente, actualizar el `image:` en `docker-compose.coolify.yml` y
redeployar. `v0.1.6` también existe (publicado 8 días antes).
- **El entrypoint vuelve a buildear `dist`** cada vez que algún env var del
prefijo `VITE_*` / `AUTH_MODE` / `POSTHOG_*` / `TURNSTILE_SITE_KEY` /
`BYPASS_EMAIL_VERIFICATION` cambie. Es intencional — el fingerprint está
ahí para no rehacer cuando nada relevante cambió.
- **Coolify normaliza el compose al guardarlo y borra los comentarios.** La
versión con explicaciones es la del repo, no la que se ve en la UI.
@@ -0,0 +1,78 @@
# OpenSEO self-host — Coolify stack (LXC 102)
#
# Por qué este compose en lugar del repo upstream (`every-app/open-seo`) vía
# railpack:
# - el deploy anterior (uuid kj0kccsb4d46tm0d6qe6docy) terminó con Caddy
# respondiendo 404 a todo porque /app/dist no existía en la imagen cacheada
# (Coolify saltó el build: "No build configuration changed & image found ...
# Build step skipped").
# - el README upstream lo dice explícitamente: "We recommend self-hosting
# with Cloudflare as opposed to Railway, Coolify or Dokploy. We plan to
# make it simpler to host on those platforms in the next few months."
# - esta imagen (`ghcr.io/every-app/open-seo:sha-c469a48`) es la build del
# mismo commit, pero el build de Vite corre en `docker-entrypoint.sh` al
# arrancar el contenedor, no en el build de la imagen. Y el entrypoint ya
# tiene la lógica de fingerprint para no reconstruir cuando los env vars
# relevantes no cambiaron.
#
# Contrato: docs/AGENTS-coolify-apps.md
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
# - SECRETOS vía env vars inyectados por Coolify (§2.5)
# - volumen con nombre para /app/.wrangler (SQLite que sobrevive al redeploy, §5)
# - healthcheck independiente de servicios externos al boot
# - AUTH_MODE=local_noauth porque aquí no hay TEAM_DOMAIN/POLICY_AUD de
# Cloudflare Access. Si más adelante se quiere proteger con auth, cambiar
# a AUTH_MODE=cloudflare_access + TEAM_DOMAIN + POLICY_AUD.
# - ALLOWED_HOST es obligatorio detrás del túnel: sin él Vite bloquea toda
# petición externa con "Blocked request" (ver preflight info level).
services:
app:
image: 'ghcr.io/every-app/open-seo:sha-c469a48'
environment:
# Puerto en el que escucha `vite preview` (per Dockerfile.selfhost / entrypoint).
- PORT=3001
# Single admin user, sin pantalla de login. NO exponer públicamente sin
# poner tu propia auth delante — el preflight lo dice literal.
- AUTH_MODE=local_noauth
# Host header permitido. Es el FQDN público por el que llega el tráfico
# desde el túnel de Cloudflare.
- ALLOWED_HOST=openseo.urieljareth.org
# Requerido por el runtime workerd para exponer process.env a los
# bindings (lo exige el compose upstream).
- CLOUDFLARE_INCLUDE_PROCESS_ENV=true
# SEO data (opcional). Vacío = la app arranca, los workflows SEO
# devuelven "no data". Se setea después vía Coolify env.
- DATAFORSEO_API_KEY=${DATAFORSEO_API_KEY:-}
# Telemetry opt-out (también vía DO_NOT_TRACK). Por defecto apagado.
- OPENSEO_TELEMETRY_DISABLED=${OPENSEO_TELEMETRY_DISABLED:-}
- DO_NOT_TRACK=${DO_NOT_TRACK:-}
# AI features (SAM, el agente SEO integrado). Vacío = SAM deshabilitado.
- OPENROUTER_API_KEY=${OPENROUTER_API_KEY:-}
- OPENROUTER_MODEL=${OPENROUTER_MODEL:-}
expose:
- '3001'
# El endpoint /api/health lo sirve el propio preflight (ver
# src/lib/selfhost-preflight.ts) sin auth — seguro para el healthcheck.
# Usamos `node` directamente porque la imagen es node:22 y el HEALTHCHECK
# upstream hace exactamente esto.
healthcheck:
test:
- CMD-SHELL
- "node -e \"fetch('http://127.0.0.1:3001/api/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))\""
interval: 30s
timeout: 10s
retries: 5
# Primer arranque: preflight + migrations + vite build (1-2 min).
start_period: 300s
volumes:
- 'openseo-data:/app/.wrangler'
restart: unless-stopped
logging:
driver: json-file
options:
max-size: '10m'
max-file: '3'
volumes:
openseo-data: {}