Actualiza toolkit operativo y documentación
This commit is contained in:
+60
-16
@@ -1,28 +1,72 @@
|
||||
# Plantilla de secretos. Copia a .env.local.ps1 (gitignored) y rellena.
|
||||
# Copy-Item .env.example .env.local.ps1
|
||||
#
|
||||
# Nota: .env.local.ps1 es un script de PowerShell que se carga con dot-source
|
||||
# (`. .\.env.local.ps1`), así que cada línea va como $env:NOMBRE = "valor".
|
||||
# Este archivo usa formato KEY=VALUE solo como referencia de qué se necesita.
|
||||
#
|
||||
# Qué necesita cada ruta: docs/TOOL-INDEX.md
|
||||
|
||||
# ── Proxmox: SSH ────────────────────────────────────────────────────────────
|
||||
# Basta con esto para todo el trabajo por SSH (el resto son los defaults del
|
||||
# repo, en scripts/ProxmoxAgent.ps1). La llave del repo (keys/proxmox_ed25519)
|
||||
# es idéntica a C:\Users\Uriel Jareth\.ssh\coolify_key — sin passphrase.
|
||||
# NOTA: la ruta antigua …\.openclaw\workspace\proxmox_key_win YA NO EXISTE.
|
||||
PROXMOX_HOST=192.168.0.200
|
||||
PROXMOX_NODE=thinkcentre
|
||||
PROXMOX_USER=root
|
||||
PROXMOX_SSH_KEY=C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win
|
||||
PROXMOX_API_BASE_URL=https://192.168.0.200:8006/api2/json
|
||||
PROXMOX_API_TOKEN_ID=root@pam!openclaw
|
||||
PROXMOX_API_TOKEN_SECRET=REPLACE_WITH_TOKEN_SECRET
|
||||
PROXMOX_SSH_KEY=keys/proxmox_ed25519
|
||||
PROXMOX_COOLIFY_LXC=102
|
||||
|
||||
# ── Proxmox: API REST ───────────────────────────────────────────────────────
|
||||
# Requerido por Invoke-ProxmoxApi. Sin ambos, lanza excepción.
|
||||
# Genera el token en la UI de Proxmox: Datacenter → Permissions → API Tokens.
|
||||
PROXMOX_API_BASE_URL=https://192.168.0.200:8006/api2/json
|
||||
PROXMOX_API_TOKEN_ID=root@pam!nombre-del-token
|
||||
PROXMOX_API_TOKEN_SECRET=REPLACE_WITH_TOKEN_SECRET
|
||||
|
||||
# ── Coolify: API ────────────────────────────────────────────────────────────
|
||||
# Requerido por coolify_skill/scripts/Invoke-CoolifyApi.ps1.
|
||||
# Instancia actual: v4.3.14 — la API está completa contra el ORIGEN
|
||||
# (http://<COOLIFY_HOST_LAN>:8000/api/v1); vía Cloudflare, /applications/* y
|
||||
# /github-apps devuelven 404 por un bloqueo del edge (no de Coolify).
|
||||
# Desde v4.2, los endpoints de estado exigen POST (p. ej. /deploy).
|
||||
COOLIFY_API_URL=https://coolify.urieljareth.org/api/v1
|
||||
COOLIFY_TOKEN=REPLACE_WITH_COOLIFY_TOKEN
|
||||
|
||||
# Used by deploy_skill (New-GitHubRepo.ps1, Invoke-GitHubApi.ps1) to create
|
||||
# repos via the GitHub REST API. git push/pull uses Windows Credential Manager
|
||||
# (wincred) and does NOT need this token in the URL.
|
||||
# Generate at https://github.com/settings/tokens (classic 'repo' scope, or
|
||||
# fine-grained with Contents:Read+Write and Metadata:Read).
|
||||
GITHUB_TOKEN=REPLACE_WITH_GITHUB_PAT
|
||||
# ── Coolify: login de la UI web ─────────────────────────────────────────────
|
||||
# Requerido SOLO por el flujo Playwright de deploy_skill/scripts/coolify-ui/,
|
||||
# que es la vía soportada para apps git build-from-source (porque la API de
|
||||
# /applications da 404). Son las credenciales con las que entras al dashboard.
|
||||
COOLIFY_EMAIL=REPLACE_WITH_COOLIFY_LOGIN_EMAIL
|
||||
COOLIFY_PASSWORD=REPLACE_WITH_COOLIFY_LOGIN_PASSWORD
|
||||
|
||||
# Used by gitea_skill (Invoke-GiteaApi.ps1, New-GiteaRepo.ps1, Sync-GiteaRemote.ps1).
|
||||
# Self-hosted Gitea behind the Cloudflare tunnel. Generate a token at:
|
||||
# <GITEA_URL>/user/settings/applications
|
||||
# Scopes needed: read:repository, write:repository, read:user (and write:admin
|
||||
# only if you manage other users). git push uses a one-shot http.extraHeader
|
||||
# injected by Sync-GiteaRemote.ps1 — the token is NOT persisted to .git/config.
|
||||
# ── Cloudflare: API ─────────────────────────────────────────────────────────
|
||||
# Requerido por scripts/Invoke-CloudflareApi.ps1 (túnel + DNS).
|
||||
# El túnel es gestionado desde el dashboard: corrige rutas ahí o por esta API,
|
||||
# nunca editando archivos en el host.
|
||||
# Genera el token en https://dash.cloudflare.com/profile/api-tokens
|
||||
CLOUDFLARE_API_TOKEN=REPLACE_WITH_CLOUDFLARE_API_TOKEN
|
||||
|
||||
# ── GitHub ──────────────────────────────────────────────────────────────────
|
||||
# Usado por deploy_skill (New-GitHubRepo.ps1, Invoke-GitHubApi.ps1) para crear
|
||||
# repos vía la API REST. `git push` usa Windows Credential Manager (wincred) y
|
||||
# NO necesita este token en la URL.
|
||||
# Un PAT aquí puede caducar (el de 2026-08 lo hizo); el token vivo de wincred
|
||||
# se recupera con:
|
||||
# printf "protocol=https\nhost=github.com\n\n" | git credential fill
|
||||
# Genera en https://github.com/settings/tokens — scope clásico 'repo', o
|
||||
# fine-grained con Contents:Read+Write y Metadata:Read.
|
||||
GITHUB_TOKEN=REPLACE_WITH_GITHUB_PAT
|
||||
GITHUB_OWNER=urieljarethbusiness-cpu
|
||||
|
||||
# ── Gitea ───────────────────────────────────────────────────────────────────
|
||||
# Usado por gitea_skill (Invoke-GiteaApi.ps1, New-GiteaRepo.ps1,
|
||||
# Sync-GiteaRemote.ps1). Gitea self-hosted detrás del túnel de Cloudflare.
|
||||
# Genera un token en <GITEA_URL>/user/settings/applications
|
||||
# Scopes: read:repository, write:repository, read:user (write:admin solo si
|
||||
# administras otros usuarios). `git push` usa un http.extraHeader de un solo uso
|
||||
# inyectado por Sync-GiteaRemote.ps1 — el token NO se persiste en .git/config.
|
||||
GITEA_URL=https://gitea-hjwh0svsoo9p5w5kj2j6b1bd.urieljareth.org
|
||||
GITEA_USER=urieljareth
|
||||
GITEA_TOKEN=REPLACE_WITH_GITEA_TOKEN
|
||||
|
||||
@@ -7,6 +7,9 @@ ACCESS.md
|
||||
*.pem
|
||||
*.pfx
|
||||
*.crt
|
||||
keys/
|
||||
.keys/
|
||||
.ssh/
|
||||
*.log
|
||||
*.tmp
|
||||
__pycache__/
|
||||
@@ -20,3 +23,9 @@ node_modules/
|
||||
.claude/
|
||||
.opencode/
|
||||
|
||||
|
||||
# Copias de rollback de composes de Coolify (pueden contener valores reales)
|
||||
backups/
|
||||
|
||||
# homelab-skill: bundle autocontenido con credenciales reales para agentes (nunca commitear)
|
||||
homelab-skill/
|
||||
|
||||
@@ -1,128 +1,193 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
Guía para Claude Code (claude.ai/code) al trabajar en este repositorio.
|
||||
|
||||
## What this repo is
|
||||
## Antes de invocar cualquier herramienta
|
||||
|
||||
This is **not an application codebase**. It is an operations toolkit: PowerShell
|
||||
wrapper scripts plus context/runbook documentation that let an agent diagnose and
|
||||
manage a single local Proxmox host (`192.168.0.200`, node `thinkcentre`) and the
|
||||
self-hosted Coolify stack running inside its LXC `102`. There is no build, lint,
|
||||
or test step — the "commands" are the operational scripts in `scripts/` and
|
||||
`coolify_skill/scripts/`.
|
||||
1. Carga los secretos: `. .\.env.local.ps1` (gitignored). Sin esto, todas las
|
||||
llamadas a API lanzan excepción.
|
||||
2. Lee **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** — es el catálogo canónico de
|
||||
todo lo ejecutable: firma real de cada script, variables de entorno que
|
||||
requiere, y si es solo-lectura o mutante. **Su §1 lista seis gotchas que
|
||||
producen resultados silenciosamente incorrectos.** No los adivines.
|
||||
3. Para el estado actual del sistema (qué existe, qué versión corre):
|
||||
[docs/proxmox-inventory.md](docs/proxmox-inventory.md).
|
||||
|
||||
User-facing docs are in Spanish; scripts and skill files are in English.
|
||||
Los seis gotchas, en una línea cada uno (detalle en el índice):
|
||||
|
||||
## Core architecture
|
||||
- **`-Raw` está invertido entre wrappers.** En Coolify y Cloudflare, sin `-Raw`
|
||||
recibes un *string*, no objetos: filtrar da vacío sin error.
|
||||
- **`Invoke-ProxmoxSsh.ps1` corrompe comillas anidadas.** Para comandos con más
|
||||
de un nivel de comillas, codifica en base64.
|
||||
- **Los nombres de contenedor llevan sufijo uuid y cambian en cada redeploy.**
|
||||
Resuélvelos siempre; nunca los escribas a mano.
|
||||
- **`/applications/*` de la API de Coolify da 404 por el hostname público
|
||||
(Cloudflare), no por Coolify.** Contra el origen
|
||||
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`) la API
|
||||
completa responde; y desde v4.2 los endpoints de estado exigen POST.
|
||||
- **"Verde en Coolify" no es "enrutado en Traefik".** La UI mira `running`;
|
||||
Traefik exige `healthy`. Un servicio nuevo aún arrancando da `503 no available
|
||||
server` sin que nada esté mal configurado.
|
||||
- **Un `502` casi siempre es el puerto.** Coolify saca el puerto de Traefik del
|
||||
`:puerto` del FQDN guardado; sin él cae al `EXPOSE` de la imagen. Si no
|
||||
coinciden, `connection refused`.
|
||||
|
||||
**Everything reaches the host through one SSH path.** There is no direct Docker or
|
||||
local network access. The layering is:
|
||||
## Qué es este repo
|
||||
|
||||
**No es el código de una aplicación.** Es un toolkit de operaciones: wrappers de
|
||||
PowerShell más documentación de contexto y runbooks que permiten a un agente
|
||||
diagnosticar y administrar un homelab — un host Proxmox local (`192.168.0.200`,
|
||||
nodo `thinkcentre`) y el stack Coolify que corre dentro de su LXC `102`. No hay
|
||||
build, lint ni tests: los "comandos" son los scripts operativos.
|
||||
|
||||
**Convención de idioma:** la documentación (`docs/`, `README.md`, este archivo)
|
||||
está en español. Los scripts y los `SKILL.md`/`TOOLS.md` de cada skill están en
|
||||
inglés.
|
||||
|
||||
## Router de intención → herramienta
|
||||
|
||||
Lo que pide el usuario, y con qué se resuelve. Para la firma completa de cada
|
||||
script, ve al [índice](docs/TOOL-INDEX.md).
|
||||
|
||||
| El usuario pide… | Usa |
|
||||
|---|---|
|
||||
| "¿está bien el servidor?", "revisa el Proxmox" | `.\scripts\Test-ProxmoxConnection.ps1` y luego `.\scripts\Get-ProxmoxInventory.ps1` |
|
||||
| "inventario de LXC/VMs", "qué hay corriendo" | `.\scripts\Get-ProxmoxInventory.ps1` |
|
||||
| "¿está corriendo X?", "estado de los contenedores" | `.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -Filter <regex>` |
|
||||
| "logs de X" | `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker logs <nombre-resuelto> --tail 100"` |
|
||||
| "qué apps hay en Coolify", "dame el uuid de X" | `.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw` |
|
||||
| cualquier cosa de la API de Coolify | `.\coolify_skill\scripts\Invoke-CoolifyApi.ps1` — pero revisa primero qué endpoints viven (§1.4 del índice) |
|
||||
| "¿está online el sitio X?" | `.\deploy_skill\scripts\Test-ServiceOnline.ps1 -Fqdn https://x.urieljareth.org` |
|
||||
| "el servicio nuevo está en verde pero da 503 / `no available server`" | `.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600` — **espera, no redeployes**; ver §1.5 del índice |
|
||||
| "arregla el túnel de Cloudflare", rutas/DNS | `.\scripts\Invoke-CloudflareApi.ps1` + [docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md) |
|
||||
| "Chatwoot perdió el enterprise" | diagnostica con `.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep`; repara con `Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures` |
|
||||
| "que arranque solo tras un apagón" | `.\scripts\Install-CoolifyAutostart.ps1 -VerifyOnly` primero |
|
||||
| "publica este proyecto en Coolify" | **lee §4 del índice antes**. Stack compose → `New-CoolifyService.ps1` (arreglado y verificado el 2026-08-24); app git → flujo UI con Playwright |
|
||||
| "una app de Coolify sigue el `main` de upstream y se rompió" | [docs/casos/firecrawl-stack-minimo.md](docs/casos/firecrawl-stack-minimo.md) — cambiar a imágenes precompiladas y compose propio |
|
||||
| "valida que este proyecto se puede deployar" | `.\deploy_skill\scripts\Test-PreDeployChecklist.ps1 -Path <ruta> -Strict` |
|
||||
| "estoy construyendo una app para este Coolify" | [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) (contrato de red, puertos, dominios, volúmenes) |
|
||||
| "crea/lista un repo en Gitea", "sube esto a Gitea" | `.\gitea_skill\scripts\` — ver §5 del índice |
|
||||
| "diagnosticar/iniciar Hermes (LXC 100)", "MiniMax-M3" | `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"` + [docs/casos/hermes-minimax-m3-setup.md](docs/casos/hermes-minimax-m3-setup.md) |
|
||||
| "un contenedor no conecta a su base de datos" | ver "Gotcha de red Docker" abajo |
|
||||
|
||||
## Arquitectura
|
||||
|
||||
**Todo llega al host por un único camino SSH.** No hay acceso directo a Docker ni
|
||||
a la red interna:
|
||||
|
||||
```
|
||||
PowerShell script → Invoke-ProxmoxSshCommand (scripts/ProxmoxAgent.ps1)
|
||||
script PowerShell → Invoke-ProxmoxSshCommand (scripts/ProxmoxAgent.ps1)
|
||||
→ ssh [email protected]
|
||||
→ pct exec 102 -- docker ... (for any Docker/Coolify container work)
|
||||
→ pct exec 102 -- docker ... (cualquier trabajo de Docker/Coolify)
|
||||
```
|
||||
|
||||
- [scripts/ProxmoxAgent.ps1](scripts/ProxmoxAgent.ps1) is the shared library. **Dot-source it** (`. .\scripts\ProxmoxAgent.ps1`) to get `Get-ProxmoxConfig`, `Invoke-ProxmoxSshCommand`, and `Invoke-ProxmoxApi`. Every other script dot-sources it rather than reimplementing connection logic.
|
||||
- **Config resolution:** `Get-ProxmoxConfig` reads env vars (`PROXMOX_HOST`, `PROXMOX_NODE`, `PROXMOX_SSH_KEY`, `PROXMOX_COOLIFY_LXC`, etc.) and falls back to hardcoded local defaults. SSH works with defaults alone; **API calls require `PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`** (otherwise `Invoke-ProxmoxApi` throws). The Coolify LXC ID (`102`) comes from config — don't hardcode it in new scripts; use `$config.CoolifyLxc`.
|
||||
- Two independent APIs: the **Proxmox REST API** (via `curl.exe -k`, PVE token header) and the **Coolify API** (via `Invoke-RestMethod`, Bearer token, base `https://coolify.urieljareth.org/api/v1`). They use different scripts and different env vars.
|
||||
- Docker is **not** managed on the Proxmox host directly — it lives inside LXC `102`. Any container command must be wrapped as `pct exec 102 -- docker ...`.
|
||||
- **The deploy pipeline** ([deploy_skill/](deploy_skill/)) is the end-to-end path from "local project" to "live on `*.urieljareth.org`": scaffold compliant Dockerfile/compose → pre-deploy checklist → create GitHub repo → push → register + deploy via Coolify API → verify. It also covers rollback. Requires `GITHUB_TOKEN` + `COOLIFY_TOKEN` in `.env.local.ps1`; `git push` uses Windows Credential Manager (`wincred`), verified for the `urieljarethbusiness-cpu` GitHub account. See [deploy_skill/SKILL.md](deploy_skill/SKILL.md) and [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md).
|
||||
- [scripts/ProxmoxAgent.ps1](scripts/ProxmoxAgent.ps1) es la librería compartida.
|
||||
**Hazle dot-source** (`. .\scripts\ProxmoxAgent.ps1`) para obtener
|
||||
`Get-ProxmoxConfig`, `Invoke-ProxmoxSshCommand` e `Invoke-ProxmoxApi`. Todos los
|
||||
demás scripts la consumen en lugar de reimplementar la conexión.
|
||||
- **Resolución de configuración:** `Get-ProxmoxConfig` lee variables de entorno
|
||||
(`PROXMOX_HOST`, `PROXMOX_NODE`, `PROXMOX_SSH_KEY`, `PROXMOX_COOLIFY_LXC`…) y
|
||||
cae a los defaults locales hardcodeados. SSH funciona solo con los defaults;
|
||||
**las llamadas a la API exigen `PROXMOX_API_TOKEN_ID` +
|
||||
`PROXMOX_API_TOKEN_SECRET`** (si faltan, `Invoke-ProxmoxApi` lanza). El ID del
|
||||
LXC de Coolify (`102`) viene de la config: usa `$config.CoolifyLxc`, no lo
|
||||
hardcodees en scripts nuevos.
|
||||
- **Cuatro APIs independientes**, con scripts y variables distintas: Proxmox REST
|
||||
(`curl.exe -k`, header de token PVE), Coolify (`Invoke-RestMethod`, Bearer,
|
||||
base `https://coolify.urieljareth.org/api/v1`), Cloudflare y Gitea.
|
||||
- Docker **no** se administra en el host Proxmox: vive dentro del LXC `102`.
|
||||
Todo comando de contenedor va envuelto en `pct exec 102 -- docker ...`.
|
||||
|
||||
## Common commands
|
||||
## Gotcha de red Docker
|
||||
|
||||
Run from the project root in PowerShell.
|
||||
Cuando una app de Coolify y su base de datos son contenedores hermanos en la
|
||||
misma red Docker, la app debe alcanzar la DB **por su nombre de servicio Docker,
|
||||
no por `localhost`** (Nextcloud, por ejemplo, usa el host `nextcloud-db`). El DNS
|
||||
entre servicios es la causa raíz habitual de los fallos de "conexión a la base de
|
||||
datos" aquí. Compruébalo con:
|
||||
|
||||
```powershell
|
||||
# Load private secrets (gitignored) — needed for any API call
|
||||
. .\.env.local.ps1
|
||||
|
||||
# Smoke test: config + SSH read + Docker sample + API auth (exits 1 on any FAIL)
|
||||
.\scripts\Test-ProxmoxConnection.ps1
|
||||
|
||||
# Full inventory snapshot (host, LXC, QEMU, Docker in LXC 102)
|
||||
.\scripts\Get-ProxmoxInventory.ps1
|
||||
|
||||
# Arbitrary read-only SSH command
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps -a"
|
||||
|
||||
# Proxmox REST API (after env token loaded)
|
||||
. .\scripts\ProxmoxAgent.ps1
|
||||
Invoke-ProxmoxApi -Path "/version"
|
||||
|
||||
# Coolify container status through LXC 102
|
||||
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All
|
||||
|
||||
# Coolify API (requires COOLIFY_TOKEN)
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/projects" -BodyJson $body
|
||||
|
||||
# Cloudflare API (requires CLOUDFLARE_API_TOKEN) — full tunnel/DNS control
|
||||
.\scripts\Invoke-CloudflareApi.ps1 -Path "/user/tokens/verify"
|
||||
|
||||
# Gitea API (requires GITEA_URL + GITEA_TOKEN) — self-hosted git hosting
|
||||
.\gitea_skill\scripts\Test-GiteaConnection.ps1
|
||||
.\gitea_skill\scripts\Get-GiteaRepo.ps1 -List
|
||||
.\gitea_skill\scripts\Invoke-GiteaApi.ps1 -Path "/user"
|
||||
.\gitea_skill\scripts\Sync-GiteaRemote.ps1 -AppPath .\my-app -CreateIfMissing -Force
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <app> getent hosts <nombre-servicio>"
|
||||
```
|
||||
|
||||
For the Coolify API reference: don't read the whole tree. Search
|
||||
`coolify_skill/references/` with `rg`, then open the single matching
|
||||
`references/ops/*.md` operation file.
|
||||
## Reglas de operación (las imponen las skills — cúmplelas)
|
||||
|
||||
## Deploying a new project to Coolify
|
||||
- **Solo-lectura primero.** El default es diagnosticar: list, status, logs,
|
||||
inspect, health checks.
|
||||
- **Confirma antes de cualquier cambio de estado.** Pregunta explícitamente antes
|
||||
de: `pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose
|
||||
up-down; deploys de Coolify o cualquier `POST`/`PUT`/`PATCH`/`DELETE`; escritura
|
||||
de variables de entorno; y todo cambio de firewall, red, storage, volumen,
|
||||
clave o token. Para acciones riesgosas: captura el estado actual y enuncia
|
||||
primero el camino de rollback.
|
||||
- **Nunca escribas secretos en el repo.** Ni tokens, passwords, claves privadas,
|
||||
cookies o secretos de token PVE — no en Markdown, no en scripts, no en logs.
|
||||
Viven solo en `.env.local.ps1` (gitignored) o en el almacén del SO. `.gitignore`
|
||||
bloquea además `ACCESS.md`, `*.key`, `*.pem`, `*.crt`. Al depurar bases de
|
||||
datos, verifica conectividad sin imprimir credenciales.
|
||||
- **Prefiere los scripts del repo** antes que cadenas de comandos manuales
|
||||
largas, y no ejecutes comandos destructivos amplios construidos desde strings
|
||||
generados.
|
||||
|
||||
End-to-end pipeline (scaffold → checklist → GitHub repo → push → Coolify app +
|
||||
deploy → verify). Requires `GITHUB_TOKEN` and `COOLIFY_TOKEN` in
|
||||
`.env.local.ps1`. `git push` uses Windows Credential Manager, not the PAT.
|
||||
## Chatwoot: el parche enterprise se degrada solo — pero un guard lo auto-repara
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
El parche no se pierde al actualizar. `Internal::CheckNewVersionsJob` hace ping
|
||||
diario a `hub.2.chatwoot.com` (a los `MD5(INSTALLATION_IDENTIFIER).hex % 1440`
|
||||
minutos pasada la medianoche UTC = **16:16 UTC** en esta instalación) y reescribe
|
||||
`INSTALLATION_PRICING_PLAN` con la respuesta del hub; después
|
||||
`ReconcilePlanConfigService` apaga los 9 feature flags premium en **todas** las
|
||||
cuentas. De ahí tres consecuencias:
|
||||
|
||||
# Full pipeline on a local project
|
||||
.\deploy_skill\scripts\Publish-ProjectToCoolify.ps1 `
|
||||
-AppPath .\my-app -AppName my-app `
|
||||
-Fqdn https://my-app.urieljareth.org `
|
||||
-Stack node -AppPort 8080 -Init
|
||||
- **El SQL de 3 filas no basta.** Los flags por cuenta viven en
|
||||
`accounts.feature_flags` (bitmask). Usa
|
||||
`.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures`;
|
||||
comprueba el estado con `.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep`.
|
||||
- **Los `UPDATE` por SQL no invalidan la caché Redis de `GlobalConfig`** (TTL de
|
||||
1 día, `V1:GLOBAL_CONFIG:*`) porque se saltan el `after_commit :clear_cache` de
|
||||
`InstallationConfig`. Cualquier parche por SQL debe llamar además a
|
||||
`GlobalConfig.clear_cache`.
|
||||
- **No bloquees `hub.2.chatwoot.com`.** Ese mismo host relaya el push móvil
|
||||
(`ChatwootHub.send_push`), activo aquí porque `FIREBASE_*` está vacío. En su
|
||||
lugar, [scripts/chatwoot-enterprise-guard.sh](scripts/chatwoot-enterprise-guard.sh)
|
||||
corre en el host Proxmox desde `/root/scripts/`, agendado por
|
||||
`/etc/cron.d/chatwoot-enterprise-guard` cada 5 minutos, y repara un revert
|
||||
detectado en ~13s (log: `/var/log/chatwoot-enterprise-guard.log`, solo escribe
|
||||
cuando actúa).
|
||||
|
||||
# Or step-by-step
|
||||
.\deploy_skill\scripts\Initialize-CoolifyProject.ps1 -Path .\my-app -Stack node
|
||||
.\deploy_skill\scripts\Test-PreDeployChecklist.ps1 -Path .\my-app -Strict
|
||||
.\deploy_skill\scripts\New-GitHubRepo.ps1 -Name my-app
|
||||
.\deploy_skill\scripts\New-CoolifyApplication.ps1 -RepoUrl https://github.com/urieljarethbusiness-cpu/my-app.git `
|
||||
-Fqdn https://my-app.urieljareth.org -PortsExposes 8080
|
||||
.\deploy_skill\scripts\Test-PostDeploy.ps1 -Fqdn https://my-app.urieljareth.org -ContainerName my-app-main
|
||||
Nunca pulses `Refresh` en `/super_admin/settings`.
|
||||
|
||||
# Rollback to a previous commit
|
||||
.\deploy_skill\scripts\Invoke-CoolifyRollback.ps1 -AppPath .\my-app -ApplicationUuid <uuid> -CommitSha <sha>
|
||||
```
|
||||
> **Estado al 2026-08-07:** corre **`chatwoot/chatwoot:v4.16.2`** (el pin
|
||||
> documentado antes era `v4.16.1`, así que el pin no sostuvo la versión). El plan
|
||||
> está en `enterprise` y una ejecución manual del guard pasa correctamente, pero
|
||||
> el log registra `ERROR: no se pudo leer INSTALLATION_PRICING_PLAN` durante la
|
||||
> ventana del update. El guard tiene el nombre del contenedor hardcodeado
|
||||
> (`postgres-c11xzy2tx2cdapm32f5b89vy`), así que **un redeploy que cambie el
|
||||
> sufijo lo deja ciego**. Análisis completo, registro de ejecución y rollback:
|
||||
> [docs/runbooks/chatwoot-update.md](docs/runbooks/chatwoot-update.md).
|
||||
|
||||
## Operating rules (enforced by the skills — follow them)
|
||||
## Túnel de Cloudflare
|
||||
|
||||
- **Read-only first.** Default to diagnostics: list, status, logs, inspect, health checks.
|
||||
- **Confirm before any state change.** Explicitly ask the user before: `pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose up/down; Coolify deploys or any `POST`/`PUT`/`PATCH`/`DELETE`; env-var writes; and any firewall/network/storage/volume/key/token change. For risky actions, capture current state and state the rollback path first.
|
||||
- **Never write secrets into the repo.** No tokens, passwords, private keys, cookies, or PVE token secrets in Markdown, scripts, or logs. Secrets live only in `.env.local.ps1` (gitignored) or the OS secret store. `.gitignore` also blocks `ACCESS.md`, `*.key`, `*.pem`, `*.crt`, etc. When debugging databases, verify connectivity without echoing credentials.
|
||||
- **Prefer repo scripts over long manual command strings**, and don't run broad destructive commands built from generated strings.
|
||||
El túnel es **gestionado desde el dashboard** (el ingress baja del edge — se ve
|
||||
como `INF Updated to new configuration version=N` en los logs de `cloudflared`).
|
||||
Corrige rutas en el dashboard o vía
|
||||
[scripts/Invoke-CloudflareApi.ps1](scripts/Invoke-CloudflareApi.ps1), **nunca**
|
||||
editando archivos en el host. Los puertos 6001/6002 deben usar `http://`. Ver
|
||||
[docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md).
|
||||
|
||||
## Docker networking gotcha
|
||||
## Dónde vive el contexto
|
||||
|
||||
When a Coolify app and its database are sibling containers on the same Docker
|
||||
network, the app must reach the DB by its **Docker service name, not `localhost`**
|
||||
(e.g. Nextcloud uses host `nextcloud-db`). Service-to-service DNS is the usual
|
||||
root cause of "database connection" failures here — check with
|
||||
`pct exec 102 -- docker exec <app> getent hosts <service-name>`.
|
||||
| Qué | Dónde |
|
||||
|---|---|
|
||||
| **Catálogo de herramientas + gotchas** | **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** |
|
||||
| Estado verificado del sistema | [docs/proxmox-inventory.md](docs/proxmox-inventory.md) |
|
||||
| Procedimientos vigentes | [docs/runbooks/](docs/runbooks/) — `conexion.md`, `diagnostico.md`, `coolify-docker.md`, `seguridad.md`, `cloudflare-tunnel.md`, `autostart-coolify.md`, `nextcloud.md`, `baserow.md`, `chatwoot-update.md` |
|
||||
| Casos resueltos paso a paso | [docs/casos/](docs/casos/) — `chatwoot-enterprise-patch.md`, `hermes-minimax-m3-setup.md`, `coolify-servicio-nuevo-503-no-available-server.md`, `firecrawl-stack-minimo.md` |
|
||||
| Incidentes archivados (**no fuente de verdad**) | [docs/incidentes/](docs/incidentes/) |
|
||||
| Contrato para *construir* una app deployable | [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) |
|
||||
| Reglas operativas por dominio | [agent/SKILL.md](agent/SKILL.md), [coolify_skill/SKILL.md](coolify_skill/SKILL.md), [deploy_skill/SKILL.md](deploy_skill/SKILL.md), [gitea_skill/SKILL.md](gitea_skill/SKILL.md) |
|
||||
| Ejemplos ejecutables por dominio | el `TOOLS.md` de cada carpeta `*_skill/` |
|
||||
| Referencia de la API de Coolify | `coolify_skill/references/ops/*.md` — **busca con `rg`, no cargues el árbol** |
|
||||
|
||||
## Where context lives
|
||||
|
||||
- [docs/proxmox-inventory.md](docs/proxmox-inventory.md) — verified topology, LXC list, observed containers, access model. Treat as the source of truth for current state.
|
||||
- [docs/runbooks/](docs/runbooks/) — concrete procedures: `conexion.md`, `diagnostico.md`, `coolify-docker.md`, `seguridad.md`, `cloudflare-tunnel.md`, `autostart-coolify.md` (power-outage auto-start of LXC 102 + tunnel), plus app-specific `nextcloud.md` and `baserow.md`.
|
||||
- Cloudflare tunnel: the tunnel is **dashboard-managed** (ingress comes from the edge, see `INF Updated to new configuration version=N` in `cloudflared` logs) — fix routes in the dashboard or via [scripts/Invoke-CloudflareApi.ps1](scripts/Invoke-CloudflareApi.ps1), not by editing files on the host. Ports 6001/6002 must use `http://`. See [docs/runbooks/cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md).
|
||||
- [agent/SKILL.md](agent/SKILL.md) + [agent/TOOLS.md](agent/TOOLS.md) — Proxmox agent operating skill.
|
||||
- [coolify_skill/SKILL.md](coolify_skill/SKILL.md) + [coolify_skill/TOOLS.md](coolify_skill/TOOLS.md) — Coolify agent operating skill and API reference index.
|
||||
- [docs/AGENTS-coolify-apps.md](docs/AGENTS-coolify-apps.md) — compatibility guide for agents **developing** an app to deploy on this Coolify instance (networking, ports, domains/TLS, env, volumes, healthchecks) + pre-deploy checklist. Source this when building a new app, not when operating an existing one.
|
||||
- [deploy_skill/SKILL.md](deploy_skill/SKILL.md) + [deploy_skill/TOOLS.md](deploy_skill/TOOLS.md) — deploy pipeline skill (scaffold → push → Coolify API → verify → rollback).
|
||||
- [gitea_skill/SKILL.md](gitea_skill/SKILL.md) + [gitea_skill/TOOLS.md](gitea_skill/TOOLS.md) — self-hosted Gitea skill: list/create/mirror repos, wire a `gitea` remote, push headlessly via one-shot `http.extraHeader` (token never persisted). Distinct from deploy_skill (which targets Coolify); this manages the git hosting layer.
|
||||
- `PROXMOX/`, `proxmox-agent/`, `proxmox-skill/` are **legacy pointer folders** — they only redirect to the live docs above. Don't add content there.
|
||||
Para la referencia de la API de Coolify: no leas el árbol completo. Busca en
|
||||
`coolify_skill/references/` con `rg` y abre el único `references/ops/*.md` que
|
||||
coincida.
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
Este archivo queda solo como puntero legacy.
|
||||
|
||||
La documentacion viva esta en:
|
||||
|
||||
- `../README.md`
|
||||
- `../agent/SKILL.md`
|
||||
- `../agent/TOOLS.md`
|
||||
- `../docs/proxmox-inventory.md`
|
||||
- `../docs/runbooks/`
|
||||
|
||||
Los secretos que antes estaban en Markdown deben vivir en variables de entorno
|
||||
o en un `.env.local.ps1` privado.
|
||||
@@ -1,80 +1,115 @@
|
||||
# Proxmox & Coolify Manager
|
||||
|
||||
Proyecto local para operar Proxmox con ayuda agentica desde Codex.
|
||||
Toolkit para operar un homelab con ayuda de un agente: un host Proxmox local y el
|
||||
stack Coolify que corre dentro de su LXC `102`, más el túnel de Cloudflare y los
|
||||
repos en GitHub y Gitea.
|
||||
|
||||
La idea practica es simple: este repo guarda el contexto, los runbooks y los
|
||||
scripts seguros para que Codex pueda diagnosticar y ayudarte a gestionar el host
|
||||
Proxmox local sin depender de memoria suelta ni de secretos pegados en Markdown.
|
||||
La idea es simple: este repo guarda el contexto, los runbooks y los scripts
|
||||
seguros para que un agente pueda diagnosticar y ayudarte a gestionar la
|
||||
infraestructura sin depender de memoria suelta ni de secretos pegados en Markdown.
|
||||
|
||||
## Empieza por aquí
|
||||
|
||||
| Si quieres… | Lee |
|
||||
|---|---|
|
||||
| saber **qué script usar** para algo | **[docs/TOOL-INDEX.md](docs/TOOL-INDEX.md)** — el catálogo canónico |
|
||||
| saber **qué existe y qué versión corre** | [docs/proxmox-inventory.md](docs/proxmox-inventory.md) |
|
||||
| que un agente opere el sistema | [CLAUDE.md](CLAUDE.md) — arquitectura + router de intención |
|
||||
| resolver un problema concreto | [docs/runbooks/](docs/runbooks/) |
|
||||
|
||||
> **Cuatro trampas conocidas** producen resultados silenciosamente incorrectos
|
||||
> (el flag `-Raw` invertido entre wrappers, comillas anidadas corrompidas en SSH,
|
||||
> nombres de contenedor no adivinables, y el 404 que Cloudflare impone a
|
||||
> `/applications/*` en el hostname público). Están documentadas en la §1 del
|
||||
> índice. Léela antes de operar.
|
||||
|
||||
## Estado verificado
|
||||
|
||||
Verificado el 2026-05-30 desde esta maquina:
|
||||
Verificado el **2026-08-29** desde esta máquina:
|
||||
|
||||
- Host Proxmox: `192.168.0.200`
|
||||
- Nodo: `thinkcentre`
|
||||
- Version: Proxmox VE `9.1.1`
|
||||
- Kernel: `6.17.2-1-pve`
|
||||
- LXC detectados: `100 hermes`, `102 coolify`
|
||||
- Docker corre dentro del LXC `102`
|
||||
- SSH con clave local funciona
|
||||
- API REST autenticada funciona cuando el token se carga desde entorno
|
||||
- Host Proxmox: `192.168.0.200`, nodo `thinkcentre`, Proxmox VE `9.1.1`,
|
||||
kernel `6.17.2-1-pve`
|
||||
- LXC: `100 hermes` (running — commit `5d3c15aaa`, MiniMax-M3), `102 coolify` (running)
|
||||
- Sin VMs QEMU
|
||||
- Docker corre dentro del LXC `102`: **68 contenedores**, 28 recursos
|
||||
registrados en Coolify
|
||||
- Coolify `v4.3.14` — su API está **completa contra el origen**
|
||||
(`http://192.168.0.117:8000/api/v1`); vía Cloudflare `/applications/*` y
|
||||
`/github-apps` devuelven 404 (bloqueo del edge, no de Coolify). Los endpoints
|
||||
de estado exigen POST desde v4.2.
|
||||
- SSH con clave local: OK · API de Coolify: OK · **API de Proxmox: OK**
|
||||
(token `root@pam!openclaw` cargado en `.env.local.ps1`)
|
||||
- API de Cloudflare: **sin token cargado**, se gestiona desde el dashboard
|
||||
|
||||
Credenciales y llaves SSH verificadas: `ACCESS.md` (local, gitignored).
|
||||
|
||||
## Estructura
|
||||
|
||||
- `agent/SKILL.md`: reglas operativas para que Codex actue como agente Proxmox.
|
||||
- `agent/TOOLS.md`: comandos seguros y patrones de uso.
|
||||
- `coolify_skill/`: skill local para operar Coolify, Docker en LXC `102` y API
|
||||
de Coolify sin guardar secretos.
|
||||
- `deploy_skill/`: skill para deployar proyectos nuevos a Coolify de punta a
|
||||
punta (scaffold Dockerfile/compose → checklist → repo GitHub → push → alta
|
||||
via API → deploy → verificacion → rollback). Necesita `GITHUB_TOKEN` y
|
||||
`COOLIFY_TOKEN` en `.env.local.ps1`.
|
||||
- `gitea_skill/`: skill para operar la instancia Gitea self-hosted (crear/listar/
|
||||
buscar repos, migrar desde GitHub, wire de un remoto `gitea` y push headless
|
||||
con token inyectado por invocacion — sin persistirlo en `.git/config`).
|
||||
Necesita `GITEA_URL`, `GITEA_USER`, `GITEA_TOKEN` en `.env.local.ps1`.
|
||||
- `docs/proxmox-inventory.md`: inventario verificado y notas de arquitectura.
|
||||
- `docs/runbooks/`: procedimientos concretos para conexion, diagnostico y Coolify.
|
||||
- `docs/runbooks/nextcloud.md`: recuperacion y fix HTTPS para Nextcloud.
|
||||
- `docs/runbooks/baserow.md`: puesta en vivo de Baserow y fix red/Traefik.
|
||||
- `docs/casos/`: casos verificados paso a paso (ej. `chatwoot-enterprise-patch.md`).
|
||||
- `docs/AGENTS-coolify-apps.md`: guia para agentes/LLMs que desarrollan una app destinada a esta instancia de Coolify (reglas de red, puertos, dominios/TLS, env, volumenes, healthchecks) + checklist pre-deploy.
|
||||
- `scripts/`: wrappers PowerShell para SSH, API e inventario.
|
||||
- `PROXMOX/`, `proxmox-agent/`, `proxmox-skill/`: carpetas legacy que ahora apuntan a la documentacion viva.
|
||||
|
||||
## Configuracion local
|
||||
|
||||
Los secretos no viven en el repo. Usa variables de entorno o un archivo privado
|
||||
ignorado por Git, por ejemplo `.env.local.ps1`.
|
||||
|
||||
Plantilla:
|
||||
|
||||
```powershell
|
||||
.\scripts\Set-ProxmoxEnv.example.ps1
|
||||
```
|
||||
CLAUDE.md Arquitectura, router de intención y reglas para el agente
|
||||
docs/
|
||||
TOOL-INDEX.md Catálogo canónico de herramientas + gotchas ← empieza aquí
|
||||
proxmox-inventory.md Estado verificado del sistema (fuente de verdad)
|
||||
AGENTS-coolify-apps.md Contrato para construir una app deployable en este Coolify
|
||||
runbooks/ Procedimientos vigentes
|
||||
casos/ Casos resueltos paso a paso
|
||||
incidentes/ Archivo histórico — NO es fuente de verdad
|
||||
scripts/ Wrappers de host: SSH, API, inventario, Cloudflare, Chatwoot
|
||||
apps/ Scripts de deploy específicos de una app
|
||||
host/ Artefactos que se despliegan en el host Proxmox
|
||||
agent/ Skill de operación del host Proxmox
|
||||
coolify_skill/ Skill de operación de Coolify + referencia de su API
|
||||
deploy_skill/ Skill de deploy de proyectos nuevos a Coolify
|
||||
gitea_skill/ Skill de la capa de hosting git self-hosted
|
||||
```
|
||||
|
||||
Prueba de conexion:
|
||||
Cada carpeta `*_skill/` tiene un `SKILL.md` (reglas operativas) y un `TOOLS.md`
|
||||
(ejemplos ejecutables).
|
||||
|
||||
### Runbooks
|
||||
|
||||
| Runbook | Qué cubre |
|
||||
|---|---|
|
||||
| [conexion.md](docs/runbooks/conexion.md) | Establecer y verificar el acceso |
|
||||
| [diagnostico.md](docs/runbooks/diagnostico.md) | Triage general del host |
|
||||
| [coolify-docker.md](docs/runbooks/coolify-docker.md) | Operar Docker dentro del LXC 102 |
|
||||
| [seguridad.md](docs/runbooks/seguridad.md) | Postura de seguridad y manejo de secretos |
|
||||
| [cloudflare-tunnel.md](docs/runbooks/cloudflare-tunnel.md) | Túnel y rutas (**gestionado desde el dashboard**) |
|
||||
| [autostart-coolify.md](docs/runbooks/autostart-coolify.md) | Auto-arranque del stack tras un corte de luz |
|
||||
| [chatwoot-update.md](docs/runbooks/chatwoot-update.md) | Actualizar Chatwoot sin perder la edición enterprise |
|
||||
| [nextcloud.md](docs/runbooks/nextcloud.md) | Recuperación y fix HTTPS |
|
||||
| [baserow.md](docs/runbooks/baserow.md) | Puesta en vivo y fix de red/Traefik |
|
||||
|
||||
## Configuración local
|
||||
|
||||
Los secretos no viven en el repo. Van en `.env.local.ps1`, ignorado por Git.
|
||||
Copia la plantilla y rellénala:
|
||||
|
||||
```powershell
|
||||
.\scripts\Test-ProxmoxConnection.ps1
|
||||
Copy-Item .env.example .env.local.ps1
|
||||
```
|
||||
|
||||
Inventario rapido:
|
||||
|
||||
```powershell
|
||||
.\scripts\Get-ProxmoxInventory.ps1
|
||||
```
|
||||
|
||||
Comando SSH puntual:
|
||||
Luego, en cada sesión:
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
.\scripts\Test-ProxmoxConnection.ps1 # smoke test; sale 1 si algo falla
|
||||
.\scripts\Get-ProxmoxInventory.ps1 # inventario completo
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list"
|
||||
```
|
||||
|
||||
## Reglas de operacion
|
||||
Para la verificación operativa con navegador (`Test-ServiceOnline.ps1`) hace
|
||||
falta Node: `npm install` una vez en la raíz del repo.
|
||||
|
||||
- Primero diagnostico de solo lectura.
|
||||
- Cambios destructivos requieren confirmacion explicita: borrar, reiniciar,
|
||||
apagar, editar red, mover discos, actualizar paquetes o modificar servicios.
|
||||
- Preferir scripts del repo antes que comandos manuales largos.
|
||||
- No registrar tokens, passwords ni claves privadas en Markdown.
|
||||
## Reglas de operación
|
||||
|
||||
- **Solo-lectura primero.** El default es diagnosticar.
|
||||
- **Confirmación explícita antes de cualquier cambio de estado**: borrar,
|
||||
reiniciar, apagar, deployar, editar red/storage, actualizar paquetes o
|
||||
modificar servicios. Para acciones riesgosas, captura el estado actual y
|
||||
enuncia el rollback antes de actuar.
|
||||
- **Prefiere los scripts del repo** antes que comandos manuales largos.
|
||||
- **Nunca registres tokens, passwords ni claves privadas en Markdown.**
|
||||
|
||||
La versión completa y vinculante está en la §6 del
|
||||
[índice de herramientas](docs/TOOL-INDEX.md).
|
||||
|
||||
+17
-8
@@ -4,18 +4,27 @@ Use this project-local skill when helping manage the local Proxmox host.
|
||||
|
||||
## Scope
|
||||
|
||||
- Host: `192.168.0.200`
|
||||
- Host: `192.168.0.200`, Proxmox VE `9.1.1`, single node
|
||||
- Node: `thinkcentre`
|
||||
- Main Docker LXC: `102` (`coolify`)
|
||||
- Secondary LXC currently observed: `100` (`hermes`)
|
||||
- Access methods: SSH first, Proxmox REST API when token env vars are present.
|
||||
- Main Docker LXC: `102` (`coolify`) — **running**, 68 containers
|
||||
- Secondary LXC: `100` (`hermes`) — **running** (commit `5d3c15aaa`, MiniMax-M3, IPs: `192.168.3.23` / `192.168.3.15`)
|
||||
- No QEMU VMs
|
||||
- Access methods: SSH first, Proxmox REST API when token env vars are present
|
||||
(they are **not** loaded today — see [`TOOLS.md`](TOOLS.md))
|
||||
|
||||
## Startup routine
|
||||
|
||||
1. Read `README.md`, `docs/proxmox-inventory.md`, and the relevant runbook.
|
||||
2. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
|
||||
3. For fresh state, run `.\scripts\Get-ProxmoxInventory.ps1`.
|
||||
4. Prefer `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "<command>"` for remote commands.
|
||||
1. Load secrets: `. .\.env.local.ps1`.
|
||||
2. Read [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — the canonical tool
|
||||
catalog. **Its §1 lists four gotchas that produce silently wrong results.**
|
||||
The one that bites hardest here: `Invoke-ProxmoxSsh.ps1` mangles nested
|
||||
quotes, so anything with two levels of quoting needs base64 (see
|
||||
[`TOOLS.md`](TOOLS.md)).
|
||||
3. Read [`docs/proxmox-inventory.md`](../docs/proxmox-inventory.md) for current
|
||||
state, plus the relevant runbook in `docs/runbooks/`.
|
||||
4. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
|
||||
5. For fresh state, run `.\scripts\Get-ProxmoxInventory.ps1`.
|
||||
6. Prefer `.\scripts\Invoke-ProxmoxSsh.ps1 -Command "<command>"` for remote commands.
|
||||
|
||||
## Safety policy
|
||||
|
||||
|
||||
+107
-15
@@ -1,26 +1,34 @@
|
||||
# Tooling
|
||||
# Tooling — Proxmox host
|
||||
|
||||
All commands assume PowerShell from the project root.
|
||||
|
||||
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — verified
|
||||
> signatures, env requirements, and read-only/mutating classification for every
|
||||
> script in the repo. This file holds the host-level usage patterns.
|
||||
|
||||
## Environment
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
```
|
||||
|
||||
If no private env file is loaded, SSH still uses the local defaults from
|
||||
Without a private env file, SSH still works from the hardcoded defaults in
|
||||
`scripts/ProxmoxAgent.ps1`. API calls require token env vars.
|
||||
|
||||
## Smoke test
|
||||
> **Status of `.env.local.ps1` (2026-08-29):** `PROXMOX_API_TOKEN_ID`
|
||||
> (`root@pam!openclaw`, verified 200 against `/version` and
|
||||
> `/cluster/resources`), `PROXMOX_API_TOKEN_SECRET`, `COOLIFY_EMAIL`,
|
||||
> `COOLIFY_PASSWORD` (probable — unverified) and a working `GITHUB_TOKEN`
|
||||
> (extracted from Windows Credential Manager after the old PAT expired) are all
|
||||
> loaded. The Proxmox REST API, the Coolify API and GitHub work. Only
|
||||
> `CLOUDFLARE_API_TOKEN` is still missing — tunnel changes go through the
|
||||
> Cloudflare dashboard.
|
||||
|
||||
## Smoke test and inventory
|
||||
|
||||
```powershell
|
||||
.\scripts\Test-ProxmoxConnection.ps1
|
||||
```
|
||||
|
||||
## Inventory
|
||||
|
||||
```powershell
|
||||
.\scripts\Get-ProxmoxInventory.ps1
|
||||
.\scripts\Test-ProxmoxConnection.ps1 # config + SSH + Docker sample + API auth; exits 1 on FAIL
|
||||
.\scripts\Get-ProxmoxInventory.ps1 # host, LXC, QEMU, Docker in LXC 102
|
||||
```
|
||||
|
||||
## SSH wrapper
|
||||
@@ -31,7 +39,49 @@ If no private env file is loaded, SSH still uses the local defaults from
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps -a"
|
||||
```
|
||||
|
||||
## API from a PowerShell session
|
||||
### ⚠️ Nested quotes get mangled — use base64
|
||||
|
||||
`Invoke-ProxmoxSshCommand` passes `$Command` as a single argument to `ssh`, and
|
||||
PowerShell 5.1 destroys embedded quotes when calling a native executable. A
|
||||
command with two levels of quoting arrives corrupted (`bash: line 1: -c: command
|
||||
not found`). Encode it instead:
|
||||
|
||||
```powershell
|
||||
$remote = @'
|
||||
pct exec 102 -- docker exec -i <db-container> bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -c "SELECT 1"'
|
||||
'@
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($remote))
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "echo $b64 | base64 -d | bash 2>&1"
|
||||
```
|
||||
|
||||
The single-quoted here-string `@'...'@` is required so PowerShell does not expand
|
||||
`$POSTGRES_PASSWORD` on the Windows side. Single-level quoting
|
||||
(`docker ps --format '{{.Names}}'`) works through the plain wrapper.
|
||||
|
||||
## Safe read-only commands
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "hostname && pveversion && uname -r && uptime"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct list && qm list"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "df -h / && free -h"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pvesh get /cluster/resources"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "systemctl --failed"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format 'table {{.Names}}\t{{.Status}}\t{{.Ports}}'"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker stats --no-stream"
|
||||
```
|
||||
|
||||
## Resolving a container name
|
||||
|
||||
Coolify names containers `<service>-<uuid>` plus an optional build suffix, so they
|
||||
are **not guessable and change when a redeploy recreates the container**:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i <app>"
|
||||
```
|
||||
|
||||
## Proxmox REST API from a PowerShell session
|
||||
|
||||
Requires `PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`; throws without them.
|
||||
|
||||
```powershell
|
||||
. .\scripts\ProxmoxAgent.ps1
|
||||
@@ -40,9 +90,51 @@ Invoke-ProxmoxApi -Path "/nodes"
|
||||
Invoke-ProxmoxApi -Path "/cluster/resources"
|
||||
```
|
||||
|
||||
## Cloudflare API — tunnel and DNS
|
||||
|
||||
Requires `CLOUDFLARE_API_TOKEN`. `GET` is read-only; everything else mutates.
|
||||
`-Raw` returns objects, the default returns a JSON string.
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-CloudflareApi.ps1 -Path "/user/tokens/verify"
|
||||
```
|
||||
|
||||
The tunnel is **dashboard-managed** — fix routes there or via this API, never by
|
||||
editing files on the host. See [`docs/runbooks/cloudflare-tunnel.md`](../docs/runbooks/cloudflare-tunnel.md).
|
||||
|
||||
## Host automation
|
||||
|
||||
```powershell
|
||||
# Power-outage auto-start of LXC 102 + tunnel. Audit without touching anything:
|
||||
.\scripts\Install-CoolifyAutostart.ps1 -VerifyOnly
|
||||
```
|
||||
|
||||
Installed on the host: `coolify-autostart.service` (systemd, **enabled**) and the
|
||||
Chatwoot guard at `/root/scripts/chatwoot-enterprise-guard.sh`, scheduled by
|
||||
`/etc/cron.d/chatwoot-enterprise-guard` every 5 minutes.
|
||||
|
||||
## Chatwoot enterprise licence
|
||||
|
||||
```powershell
|
||||
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep # read-only diagnosis
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun -ReenableAccountFeatures # preview the repair
|
||||
```
|
||||
|
||||
Full context: [`docs/runbooks/chatwoot-update.md`](../docs/runbooks/chatwoot-update.md).
|
||||
|
||||
## Per-app deploy helpers
|
||||
|
||||
`scripts/apps/` holds app-specific one-off deploy scripts with hardcoded uuids
|
||||
and domains — e.g. `Deploy-SoloLeveling.ps1`, which builds the image on the
|
||||
server to work around a private GHCR without `read:packages`. Read the header
|
||||
before running one; they are **mutating** and tied to a specific resource.
|
||||
|
||||
## Commands that require confirmation
|
||||
|
||||
- `pct start`, `pct shutdown`, `pct reboot`, `pct stop`, `pct destroy`
|
||||
- `qm start`, `qm shutdown`, `qm reboot`, `qm stop`, `qm destroy`
|
||||
- `docker restart`, `docker stop`, `docker rm`, `docker compose up/down`
|
||||
- package updates, firewall edits, network edits, storage edits
|
||||
- `pct start|shutdown|reboot|stop|destroy`
|
||||
- `qm start|shutdown|reboot|stop|destroy`
|
||||
- `docker restart|stop|rm|compose up|compose down`
|
||||
- Package updates, firewall edits, network edits, storage edits
|
||||
- Any Cloudflare write (`POST`/`PUT`/`PATCH`/`DELETE`)
|
||||
- `Install-CoolifyAutostart.ps1` without `-VerifyOnly`
|
||||
- `Apply-ChatwootEnterprisePatch.ps1` without `-DryRun`
|
||||
|
||||
+20
-8
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: coolify-agent
|
||||
description: Operate the local self-hosted Coolify stack for this Proxmox and Coolify Manager project. Use when Codex needs to inspect Coolify, Docker containers inside LXC 102, applications, services, databases, deployments, logs, environment variables, or Coolify API resources on the local infrastructure.
|
||||
description: Operate the local self-hosted Coolify stack for this Proxmox and Coolify Manager project. Use when the agent needs to inspect Coolify, Docker containers inside LXC 102, applications, services, databases, deployments, logs, environment variables, or Coolify API resources on the local infrastructure.
|
||||
---
|
||||
|
||||
# Coolify Agent
|
||||
@@ -11,14 +11,18 @@ changing production state.
|
||||
|
||||
## Startup Routine
|
||||
|
||||
1. Read `README.md`, `docs/proxmox-inventory.md`, and
|
||||
`docs/runbooks/coolify-docker.md`.
|
||||
2. For app-specific work, also read the matching runbook, for example
|
||||
1. Load secrets: `. .\.env.local.ps1`.
|
||||
2. Read [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — the canonical tool
|
||||
catalog. **Its §1 lists four gotchas that produce silently wrong results**;
|
||||
two of them (`-Raw` inversion, `/applications` 404) bite on every Coolify task.
|
||||
3. Read [`docs/proxmox-inventory.md`](../docs/proxmox-inventory.md) for current
|
||||
state, and `docs/runbooks/coolify-docker.md` for the procedure.
|
||||
4. For app-specific work, also read the matching runbook, for example
|
||||
`docs/runbooks/nextcloud.md`.
|
||||
3. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
|
||||
4. Discover current state before acting:
|
||||
5. Run `.\scripts\Test-ProxmoxConnection.ps1` before operational work.
|
||||
6. Discover current state before acting:
|
||||
`.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All`.
|
||||
5. Use the Coolify API only when `COOLIFY_TOKEN` is loaded in the shell.
|
||||
7. Use the Coolify API only when `COOLIFY_TOKEN` is loaded in the shell.
|
||||
|
||||
## Local Context
|
||||
|
||||
@@ -27,7 +31,15 @@ changing production state.
|
||||
- Docker is not managed directly on Proxmox; use
|
||||
`pct exec 102 -- docker ...` through `.\scripts\Invoke-ProxmoxSsh.ps1`.
|
||||
- Default Coolify API base URL:
|
||||
`https://coolify.urieljareth.org/api/v1`.
|
||||
`https://coolify.urieljareth.org/api/v1`. Coolify runs **v4.3.14** (verified
|
||||
2026-08-29) and its REST API is **complete against the origin**
|
||||
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`): the
|
||||
`/applications/*` and `/github-apps` 404s only happen through the public
|
||||
Cloudflare hostname — an edge block, not a Coolify limitation. Call those
|
||||
endpoints against the origin. State-changing endpoints are POST-only since
|
||||
v4.2. Otherwise enumerate apps via `/resources`.
|
||||
- **Container names are `<service>-<uuid>`** and change when a redeploy recreates
|
||||
the container. Never hardcode one — resolve it first.
|
||||
- Secrets must live in local environment files or the OS secret store, never in
|
||||
Markdown or skill references.
|
||||
|
||||
|
||||
+79
-6
@@ -2,6 +2,33 @@
|
||||
|
||||
All commands assume PowerShell from the project root.
|
||||
|
||||
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md). This file
|
||||
> holds usage examples; the index holds verified signatures, env requirements,
|
||||
> read-only/mutating classification and the gotchas. Read its §1 first.
|
||||
|
||||
## Two gotchas specific to this wrapper
|
||||
|
||||
**1. `-Raw` is inverted.** `Invoke-CoolifyApi.ps1` returns a **JSON string** by
|
||||
default and **PowerShell objects with `-Raw`**. Filtering the default output
|
||||
silently yields nothing — no error:
|
||||
|
||||
```powershell
|
||||
# WRONG: $r is a System.String, so this prints an empty row
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" | Select-Object name, uuid
|
||||
|
||||
# RIGHT
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw | Select-Object name, uuid
|
||||
```
|
||||
|
||||
**2. `/applications/*` 404s through the public hostname — that block is
|
||||
Cloudflare's, not Coolify's** (re-verified 2026-08-29 on v4.3.14: same token,
|
||||
same route → 200 against the origin `http://192.168.0.117:8000/api/v1`, 404 via
|
||||
`https://coolify.urieljareth.org`). For those endpoints use
|
||||
`$env:COOLIFY_API_URL_ORIGIN`. Also: state-changing endpoints are POST-only
|
||||
since v4.2 (`GET /deploy` → 405). Endpoints that work through either path:
|
||||
`/version`, `/resources`, `/services`, `/databases`, `/projects`, `/servers`,
|
||||
`/teams`, `/deployments`. Use `/resources` to enumerate apps.
|
||||
|
||||
## Environment
|
||||
|
||||
Load private values from an ignored local file:
|
||||
@@ -31,6 +58,29 @@ $env:COOLIFY_TOKEN = "REPLACE_WITH_TOKEN"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker volume ls"
|
||||
```
|
||||
|
||||
## A New Service Is Green In Coolify But Its Domain Returns 503
|
||||
|
||||
Coolify's green dot means the container is `running`. Traefik only routes a
|
||||
container Docker reports `healthy`. A brand-new service is `running` long before
|
||||
it is `healthy`, so its route does not exist yet and the request falls through to
|
||||
Coolify's catch-all (`noop` service, empty server list) — which is what prints
|
||||
`no available server`.
|
||||
|
||||
```powershell
|
||||
# Contrast running vs healthy, flag a too-short start_period, probe the domain.
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <resource-uuid>
|
||||
|
||||
# First boot on this host can take minutes (HDD-backed loopback rootfs). Wait it out.
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <resource-uuid> -WaitSeconds 600
|
||||
```
|
||||
|
||||
Read the status code before changing anything:
|
||||
|
||||
- `502 Bad Gateway` — the route exists, the backend refuses. App or port problem.
|
||||
- `503 no available server` — there is no route. Usually still starting.
|
||||
**Do not redeploy**: that restarts the entrypoint from scratch and restarts the
|
||||
slow boot. See TOOL-INDEX.md 1.5.
|
||||
|
||||
## Logs And Inspect
|
||||
|
||||
```powershell
|
||||
@@ -40,14 +90,37 @@ $env:COOLIFY_TOKEN = "REPLACE_WITH_TOKEN"
|
||||
|
||||
## Coolify API
|
||||
|
||||
Verified working on this instance (2026-08-07):
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/version"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/databases"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" # 28 — the app inventory
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects" # 7
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services" # 14
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/databases" # 5
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers" # 1
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/teams" # 1
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments" # in-flight deploys
|
||||
```
|
||||
|
||||
**`/applications` and its whole namespace 404 through the public hostname
|
||||
(Cloudflare edge block) but work via the origin** — point
|
||||
`COOLIFY_API_URL` at `http://192.168.0.117:8000/api/v1` for those calls. Either
|
||||
way, to enumerate applications and resolve a name or uuid, `/resources` is the
|
||||
reliable inventory:
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
|
||||
Where-Object { $_.name -match 'chatwoot' } | Select-Object name, uuid, fqdn
|
||||
```
|
||||
|
||||
### Resolving a container name from a resource
|
||||
|
||||
Container names are `<service>-<uuid>` plus an optional build suffix, so they are
|
||||
not guessable and **change when a redeploy recreates the container**:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i <app>"
|
||||
```
|
||||
|
||||
For write calls, prepare the JSON body first and ask for confirmation:
|
||||
|
||||
@@ -47,8 +47,12 @@ if ($PSBoundParameters.ContainsKey("BodyJson")) {
|
||||
throw "BodyJson is not valid JSON: $($_.Exception.Message)"
|
||||
}
|
||||
|
||||
$request.Body = $BodyJson
|
||||
$request.ContentType = "application/json"
|
||||
# Send bytes, not a string. PowerShell 5.1 encodes a string body using the
|
||||
# default codepage, so any non-ASCII character (an accented comment inside a
|
||||
# compose file, for instance) reaches Coolify mangled and it answers
|
||||
# 400 {"message":"Invalid request.","error":"Invalid JSON."}.
|
||||
$request.Body = [Text.Encoding]::UTF8.GetBytes($BodyJson)
|
||||
$request.ContentType = "application/json; charset=utf-8"
|
||||
}
|
||||
|
||||
try {
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Give a Coolify service's healthchecks a start_period long enough for a first
|
||||
boot on this host. Dry-run by default.
|
||||
|
||||
.DESCRIPTION
|
||||
Coolify's library templates ship healthchecks tuned for SSD hosts: a short
|
||||
interval, a handful of retries and no start_period at all. On this host a
|
||||
first boot takes minutes (rootfs is ext4 over loopback over an HDD, ~39 ms
|
||||
per write), so the container is flagged `unhealthy` long before the app
|
||||
listens. Traefik only routes containers Docker reports `healthy`, so an
|
||||
unhealthy container has NO route and the request falls through to Coolify's
|
||||
catch-all (`noop`, empty server list) -> 503 "no available server".
|
||||
|
||||
Worse, once flagged unhealthy the container is a candidate for recreation,
|
||||
and recreating restarts the slow entrypoint from zero. That is what turns a
|
||||
transient 503 into a permanent one.
|
||||
|
||||
This script edits `services.docker_compose_raw` (the editable template
|
||||
Coolify regenerates the deployed compose from) and inserts a `start_period`
|
||||
into every healthcheck that lacks one, optionally raising a very short
|
||||
`interval`.
|
||||
|
||||
It does NOT redeploy. The new healthcheck only takes effect when the
|
||||
container is recreated, which is a separate, explicit step.
|
||||
|
||||
Honest scope: start_period does NOT make the site answer sooner. During
|
||||
startup Docker reports `starting`, which Traefik does not route either, so
|
||||
the startup 503 window still exists. What it prevents is the container being
|
||||
*marked failed* and entering the recreation loop.
|
||||
|
||||
.PARAMETER Uuid
|
||||
Coolify service uuid (last path segment of the service URL in the UI).
|
||||
|
||||
.PARAMETER StartPeriodSeconds
|
||||
Grace window to insert. Default 300 (measured first boots here ran into the
|
||||
low minutes).
|
||||
|
||||
.PARAMETER MinIntervalSeconds
|
||||
Raise any `interval` below this. A 2 s interval spawns a health exec every
|
||||
two seconds against an already saturated disk. Default 10. Pass 0 to leave
|
||||
every interval untouched.
|
||||
|
||||
.PARAMETER Apply
|
||||
Actually write. Without it the script only prints the diff and changes
|
||||
nothing.
|
||||
|
||||
.PARAMETER ShowResult
|
||||
Also print the resulting healthcheck blocks so the exact YAML can be
|
||||
reviewed before writing.
|
||||
|
||||
.EXAMPLE
|
||||
# Inspect what would change. Safe, read-only.
|
||||
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid uyn0js6pqbwo8mubw5edy95f
|
||||
|
||||
.EXAMPLE
|
||||
# Write it, then redeploy that service yourself from the Coolify UI.
|
||||
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid uyn0js6pqbwo8mubw5edy95f -Apply
|
||||
#>
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[Parameter(Mandatory = $true)]
|
||||
[string]$Uuid,
|
||||
|
||||
[int]$StartPeriodSeconds = 300,
|
||||
|
||||
[int]$MinIntervalSeconds = 10,
|
||||
|
||||
[switch]$Apply,
|
||||
|
||||
[switch]$ShowResult
|
||||
)
|
||||
|
||||
$ErrorActionPreference = "Stop"
|
||||
|
||||
$repoRoot = Resolve-Path (Join-Path $PSScriptRoot "..\..")
|
||||
$invokeSsh = Join-Path $repoRoot "scripts\Invoke-ProxmoxSsh.ps1"
|
||||
$agentScript = Join-Path $repoRoot "scripts\ProxmoxAgent.ps1"
|
||||
|
||||
foreach ($required in @($invokeSsh, $agentScript)) {
|
||||
if (-not (Test-Path -LiteralPath $required)) { throw "Missing dependency: $required" }
|
||||
}
|
||||
|
||||
. $agentScript
|
||||
$config = Get-ProxmoxConfig
|
||||
$lxc = $config.CoolifyLxc
|
||||
|
||||
# Nested quoting is corrupted by the SSH wrapper (TOOL-INDEX.md 1.2); base64
|
||||
# every remote command. Compose bodies also travel base64 so that newlines,
|
||||
# quotes and Coolify's ${...} magic survive intact.
|
||||
function Invoke-InLxc {
|
||||
param([Parameter(Mandatory = $true)][string]$Script)
|
||||
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($Script))
|
||||
return @(& $invokeSsh -Command "pct exec $lxc -- bash -c 'echo $b64 | base64 -d | bash'")
|
||||
}
|
||||
|
||||
function Invoke-CoolifyDb {
|
||||
param([Parameter(Mandatory = $true)][string]$Sql)
|
||||
|
||||
# psql -At: unaligned, no header. Quotes are safe inside the base64 payload.
|
||||
return Invoke-InLxc -Script "docker exec coolify-db psql -U coolify -At -c ""$Sql"""
|
||||
}
|
||||
|
||||
function Get-ComposeRaw {
|
||||
param([string]$ServiceUuid)
|
||||
|
||||
$sql = "select encode(convert_to(docker_compose_raw,'UTF8'),'base64') from services where uuid='$ServiceUuid'"
|
||||
$lines = Invoke-CoolifyDb -Sql $sql
|
||||
$payload = ($lines -join '').Trim()
|
||||
if (-not $payload) { return $null }
|
||||
return [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String($payload))
|
||||
}
|
||||
|
||||
function Set-ComposeRaw {
|
||||
param([string]$ServiceUuid, [string]$Content)
|
||||
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($Content))
|
||||
$sql = "update services set docker_compose_raw = convert_from(decode('$b64','base64'),'UTF8'), updated_at = now() where uuid='$ServiceUuid'"
|
||||
$out = Invoke-CoolifyDb -Sql $sql
|
||||
return ($out -join ' ').Trim()
|
||||
}
|
||||
|
||||
function Get-Indent {
|
||||
param([string]$Line)
|
||||
if ($Line -match '^(\s*)') { return $Matches[1].Length }
|
||||
return 0
|
||||
}
|
||||
|
||||
<#
|
||||
Insert start_period into every healthcheck block that lacks one, and raise a
|
||||
too-short interval. Deliberately line-based: re-serialising the YAML would
|
||||
reformat Coolify's magic placeholders and its `- SERVICE_URL_X` shorthand.
|
||||
#>
|
||||
function Update-Healthchecks {
|
||||
param(
|
||||
[string[]]$Lines,
|
||||
[int]$StartPeriod,
|
||||
[int]$MinInterval
|
||||
)
|
||||
|
||||
$out = New-Object 'System.Collections.Generic.List[string]'
|
||||
$changes = New-Object 'System.Collections.Generic.List[object]'
|
||||
|
||||
$i = 0
|
||||
while ($i -lt $Lines.Count) {
|
||||
$line = $Lines[$i]
|
||||
|
||||
if ($line -notmatch '^\s*healthcheck:\s*$') {
|
||||
$out.Add($line)
|
||||
$i++
|
||||
continue
|
||||
}
|
||||
|
||||
$hcIndent = Get-Indent -Line $line
|
||||
$out.Add($line)
|
||||
$hcLineNumber = $i + 1
|
||||
$i++
|
||||
|
||||
# Collect the block: every following line indented deeper than
|
||||
# `healthcheck:` itself. Blank lines inside the block are kept.
|
||||
$block = New-Object 'System.Collections.Generic.List[string]'
|
||||
while ($i -lt $Lines.Count) {
|
||||
$candidate = $Lines[$i]
|
||||
if ($candidate.Trim() -eq '') { $block.Add($candidate); $i++; continue }
|
||||
if ((Get-Indent -Line $candidate) -le $hcIndent) { break }
|
||||
$block.Add($candidate)
|
||||
$i++
|
||||
}
|
||||
|
||||
# Trailing blank lines belong after the block, not inside it.
|
||||
while ($block.Count -gt 0 -and $block[$block.Count - 1].Trim() -eq '') {
|
||||
$block.RemoveAt($block.Count - 1)
|
||||
$i--
|
||||
}
|
||||
|
||||
$childIndent = ' ' * ($hcIndent + 2)
|
||||
foreach ($b in $block) {
|
||||
if ($b.Trim() -ne '') { $childIndent = ' ' * (Get-Indent -Line $b); break }
|
||||
}
|
||||
|
||||
$hasStartPeriod = @($block | Where-Object { $_ -match '^\s*start_period\s*:' }).Count -gt 0
|
||||
|
||||
# Raise a too-short interval.
|
||||
for ($j = 0; $j -lt $block.Count; $j++) {
|
||||
if ($MinInterval -le 0) { break }
|
||||
if ($block[$j] -notmatch '^(\s*)interval\s*:\s*(\S+)\s*$') { continue }
|
||||
|
||||
$indent = $Matches[1]
|
||||
$current = $Matches[2]
|
||||
$seconds = $null
|
||||
if ($current -match '^(\d+(?:\.\d+)?)s$') { $seconds = [double]$Matches[1] }
|
||||
elseif ($current -match '^(\d+)$') { $seconds = [double]$Matches[1] }
|
||||
|
||||
if ($null -ne $seconds -and $seconds -lt $MinInterval) {
|
||||
$block[$j] = "${indent}interval: ${MinInterval}s"
|
||||
$changes.Add([pscustomobject]@{
|
||||
Line = $hcLineNumber
|
||||
Kind = 'interval'
|
||||
From = "interval: $current"
|
||||
To = "interval: ${MinInterval}s"
|
||||
})
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
if (-not $hasStartPeriod) {
|
||||
$block.Add("${childIndent}start_period: ${StartPeriod}s")
|
||||
$changes.Add([pscustomobject]@{
|
||||
Line = $hcLineNumber
|
||||
Kind = 'start_period'
|
||||
From = '(absent)'
|
||||
To = "start_period: ${StartPeriod}s"
|
||||
})
|
||||
}
|
||||
|
||||
foreach ($b in $block) { $out.Add($b) }
|
||||
}
|
||||
|
||||
return [pscustomobject]@{
|
||||
Lines = $out.ToArray()
|
||||
Changes = $changes.ToArray()
|
||||
}
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "Healthcheck grace - service $Uuid" -ForegroundColor Cyan
|
||||
Write-Host ("-" * 72)
|
||||
|
||||
$original = Get-ComposeRaw -ServiceUuid $Uuid
|
||||
if ($null -eq $original) {
|
||||
throw "No service with uuid '$Uuid' (or its docker_compose_raw is empty). Has it been deleted? Check: Invoke-CoolifyApi.ps1 -Path /services -Raw"
|
||||
}
|
||||
|
||||
$originalLines = $original -split "`r?`n"
|
||||
$result = Update-Healthchecks -Lines $originalLines -StartPeriod $StartPeriodSeconds -MinInterval $MinIntervalSeconds
|
||||
|
||||
$hcCount = @($originalLines | Where-Object { $_ -match '^\s*healthcheck:\s*$' }).Count
|
||||
Write-Host "healthcheck blocks found: $hcCount"
|
||||
|
||||
if ($result.Changes.Count -eq 0) {
|
||||
Write-Host "Nothing to change: every healthcheck already has a start_period and an acceptable interval." -ForegroundColor Green
|
||||
return
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "Proposed changes:" -ForegroundColor Yellow
|
||||
$result.Changes | Format-Table Line, Kind, From, To -AutoSize
|
||||
|
||||
$updated = ($result.Lines -join "`n")
|
||||
|
||||
# Guard: the edit must only ever add/modify healthcheck lines. If the line count
|
||||
# moved by more than the number of inserted lines, something went wrong.
|
||||
$inserted = @($result.Changes | Where-Object { $_.Kind -eq 'start_period' }).Count
|
||||
$delta = $result.Lines.Count - $originalLines.Count
|
||||
if ($delta -ne $inserted) {
|
||||
throw "Refusing to write: line count moved by $delta but only $inserted lines should have been inserted. The block parser mis-scoped a healthcheck."
|
||||
}
|
||||
|
||||
if ($ShowResult) {
|
||||
Write-Host ""
|
||||
Write-Host "Resulting healthcheck blocks:" -ForegroundColor Cyan
|
||||
$lines = $result.Lines
|
||||
for ($k = 0; $k -lt $lines.Count; $k++) {
|
||||
if ($lines[$k] -notmatch '^\s*healthcheck:\s*$') { continue }
|
||||
$indent = ($lines[$k] -replace '\S.*$', '').Length
|
||||
Write-Host (" {0,4}: {1}" -f ($k + 1), $lines[$k]) -ForegroundColor DarkGray
|
||||
for ($m = $k + 1; $m -lt $lines.Count; $m++) {
|
||||
if ($lines[$m].Trim() -ne '' -and (($lines[$m] -replace '\S.*$', '').Length -le $indent)) { break }
|
||||
$colour = if ($lines[$m] -match 'start_period|interval') { 'Green' } else { 'DarkGray' }
|
||||
Write-Host (" {0,4}: {1}" -f ($m + 1), $lines[$m]) -ForegroundColor $colour
|
||||
}
|
||||
Write-Host ""
|
||||
}
|
||||
}
|
||||
|
||||
if (-not $Apply) {
|
||||
Write-Host "DRY RUN - nothing was written. Re-run with -Apply to persist." -ForegroundColor Cyan
|
||||
Write-Host "After applying you must redeploy the service for it to take effect." -ForegroundColor Cyan
|
||||
return
|
||||
}
|
||||
|
||||
$backupDir = Join-Path $repoRoot "backups"
|
||||
if (-not (Test-Path -LiteralPath $backupDir)) {
|
||||
New-Item -ItemType Directory -Path $backupDir | Out-Null
|
||||
}
|
||||
$stamp = Get-Date -Format 'yyyyMMdd-HHmmss'
|
||||
$backupFile = Join-Path $backupDir "compose-raw_${Uuid}_$stamp.yml"
|
||||
[IO.File]::WriteAllText($backupFile, $original, (New-Object Text.UTF8Encoding($false)))
|
||||
Write-Host "Rollback copy: $backupFile" -ForegroundColor DarkGray
|
||||
|
||||
$status = Set-ComposeRaw -ServiceUuid $Uuid -Content $updated
|
||||
Write-Host "psql: $status"
|
||||
|
||||
# Read back and compare, rather than trusting the UPDATE.
|
||||
$verify = Get-ComposeRaw -ServiceUuid $Uuid
|
||||
if ($verify -ne $updated) {
|
||||
Write-Host "VERIFY FAILED - stored content does not match what was sent." -ForegroundColor Red
|
||||
Write-Host "Restore with the rollback copy above before doing anything else." -ForegroundColor Red
|
||||
throw "Write-back verification failed for service $Uuid."
|
||||
}
|
||||
|
||||
Write-Host "Verified: stored docker_compose_raw matches the intended content." -ForegroundColor Green
|
||||
Write-Host ""
|
||||
Write-Host "NOT redeployed. The healthcheck changes only apply once the container" -ForegroundColor Yellow
|
||||
Write-Host "is recreated. Redeploy the service from the Coolify UI, then confirm:" -ForegroundColor Yellow
|
||||
Write-Host " .\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid $Uuid -WaitSeconds 600" -ForegroundColor Yellow
|
||||
@@ -0,0 +1,270 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Read-only readiness probe for a Coolify service. Explains a 503
|
||||
"no available server" instead of leaving you guessing.
|
||||
|
||||
.DESCRIPTION
|
||||
Coolify's UI reports a service as green when its container is *running*.
|
||||
Traefik, however, only puts a container in the load balancer once Docker
|
||||
reports it *healthy*. On this host a first boot can take minutes (HDD-backed
|
||||
loopback storage, ~39 ms/write), so a brand-new service is Running but not
|
||||
yet healthy — Traefik has no route for it, the request falls through to
|
||||
Coolify's catch-all router (priority -1000, service `noop`, empty server
|
||||
list) and Traefik answers 503 "no available server".
|
||||
|
||||
This script reports both signals side by side so the gap is visible, and
|
||||
tells you whether you should simply wait.
|
||||
|
||||
Read-only: it never restarts, redeploys or mutates anything.
|
||||
|
||||
.PARAMETER Uuid
|
||||
Coolify resource UUID (the last path segment of the service URL in the UI).
|
||||
Container names carry this as a suffix and change on every redeploy, so the
|
||||
real name is resolved here rather than typed by hand.
|
||||
|
||||
.PARAMETER Fqdn
|
||||
Public URL to probe. Defaults to whatever COOLIFY_FQDN the container carries.
|
||||
|
||||
.PARAMETER WaitSeconds
|
||||
Poll until every container is healthy, up to this many seconds. Default 0
|
||||
(report once and exit). Use 600 for a first boot on this host.
|
||||
|
||||
.EXAMPLE
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid znpmxv2o6ggooi6qxksiagke
|
||||
|
||||
.EXAMPLE
|
||||
# First boot of a service from the Coolify library: wait it out.
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid znpmxv2o6ggooi6qxksiagke -WaitSeconds 600
|
||||
#>
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[Parameter(Mandatory = $true)]
|
||||
[string]$Uuid,
|
||||
|
||||
[string]$Fqdn,
|
||||
|
||||
[int]$WaitSeconds = 0
|
||||
)
|
||||
|
||||
$ErrorActionPreference = "Stop"
|
||||
|
||||
$repoRoot = Resolve-Path (Join-Path $PSScriptRoot "..\..")
|
||||
$invokeSsh = Join-Path $repoRoot "scripts\Invoke-ProxmoxSsh.ps1"
|
||||
$agentScript = Join-Path $repoRoot "scripts\ProxmoxAgent.ps1"
|
||||
|
||||
foreach ($required in @($invokeSsh, $agentScript)) {
|
||||
if (-not (Test-Path -LiteralPath $required)) {
|
||||
throw "Missing dependency: $required"
|
||||
}
|
||||
}
|
||||
|
||||
. $agentScript
|
||||
$config = Get-ProxmoxConfig
|
||||
$lxc = $config.CoolifyLxc
|
||||
|
||||
# Nested quoting is corrupted by the SSH wrapper (TOOL-INDEX.md 1.2), so every
|
||||
# non-trivial remote command is base64-encoded.
|
||||
function Invoke-InLxc {
|
||||
param([Parameter(Mandatory = $true)][string]$Script)
|
||||
|
||||
$bytes = [Text.Encoding]::UTF8.GetBytes($Script)
|
||||
$b64 = [Convert]::ToBase64String($bytes)
|
||||
return @(& $invokeSsh -Command "pct exec $lxc -- bash -c 'echo $b64 | base64 -d | bash'")
|
||||
}
|
||||
|
||||
function Get-ServiceContainers {
|
||||
param([string]$ResourceUuid)
|
||||
|
||||
# Container names carry the uuid as a suffix and change on every redeploy.
|
||||
$lines = Invoke-InLxc -Script @"
|
||||
docker ps -a --filter "label=coolify.resourceName" --format '{{.Names}}' 2>/dev/null | grep -- '$ResourceUuid' || true
|
||||
docker ps -a --format '{{.Names}}' 2>/dev/null | grep -- '$ResourceUuid' || true
|
||||
"@
|
||||
return @($lines | Where-Object { $_ -and $_.Trim() } | ForEach-Object { $_.Trim() } | Sort-Object -Unique)
|
||||
}
|
||||
|
||||
function Get-ContainerReport {
|
||||
param([string]$Name)
|
||||
|
||||
$raw = Invoke-InLxc -Script @"
|
||||
docker inspect '$Name' --format '{{.State.Status}}|{{if .State.Health}}{{.State.Health.Status}}|{{.State.Health.FailingStreak}}{{else}}none|0{{end}}|{{.State.ExitCode}}|{{.State.StartedAt}}|{{.RestartCount}}|{{index .Config.Labels "coolify.serviceName"}}'
|
||||
docker inspect '$Name' --format 'HC|{{if .Config.Healthcheck}}{{.Config.Healthcheck.Interval}}|{{.Config.Healthcheck.Retries}}|{{.Config.Healthcheck.StartPeriod}}{{else}}absent|0|0{{end}}'
|
||||
echo "ENVFQDN|`$(docker inspect '$Name' --format '{{range .Config.Env}}{{println .}}{{end}}' 2>/dev/null | sed -n 's/^COOLIFY_FQDN=//p' | head -1)"
|
||||
"@
|
||||
|
||||
$stateLine = @($raw | Where-Object { $_ -and $_ -notmatch '^(HC|ENVFQDN)\|' })[0]
|
||||
$hcLine = @($raw | Where-Object { $_ -match '^HC\|' })[0]
|
||||
$envLine = @($raw | Where-Object { $_ -match '^ENVFQDN\|' })[0]
|
||||
|
||||
if (-not $stateLine) { return $null }
|
||||
$f = $stateLine.Split('|')
|
||||
|
||||
$interval = $null; $retries = $null; $startPeriod = $null
|
||||
if ($hcLine) {
|
||||
$h = $hcLine.Split('|')
|
||||
$interval = $h[1]; $retries = $h[2]; $startPeriod = $h[3]
|
||||
}
|
||||
|
||||
$containerFqdn = $null
|
||||
if ($envLine) { $containerFqdn = $envLine.Split('|', 2)[1] }
|
||||
|
||||
# Docker prints healthcheck durations either as raw nanoseconds or as a Go
|
||||
# duration string ("2s", "1m30s"), depending on the daemon version.
|
||||
$toSeconds = {
|
||||
param($value)
|
||||
if (-not $value) { return $null }
|
||||
if ($value -match '^\d+$') { return [math]::Round([double]$value / 1e9, 1) }
|
||||
$total = 0.0; $matched = $false
|
||||
foreach ($m in [regex]::Matches($value, '([\d.]+)(h|ms|m|s)')) {
|
||||
$n = [double]$m.Groups[1].Value
|
||||
switch ($m.Groups[2].Value) {
|
||||
'h' { $total += $n * 3600 }
|
||||
'm' { $total += $n * 60 }
|
||||
's' { $total += $n }
|
||||
'ms' { $total += $n / 1000 }
|
||||
}
|
||||
$matched = $true
|
||||
}
|
||||
if ($matched) { return [math]::Round($total, 1) }
|
||||
return $null
|
||||
}
|
||||
|
||||
return [pscustomobject]@{
|
||||
Name = $Name
|
||||
State = $f[0]
|
||||
Health = $f[1]
|
||||
FailingStreak = [int]$f[2]
|
||||
ExitCode = $f[3]
|
||||
StartedAt = $f[4]
|
||||
RestartCount = $f[5]
|
||||
ServiceName = $f[6]
|
||||
HealthInterval = & $toSeconds $interval
|
||||
HealthRetries = $retries
|
||||
HealthStartPeriod = & $toSeconds $startPeriod
|
||||
Fqdn = $containerFqdn
|
||||
# A container that exited 0 and has no healthcheck is a one-shot init
|
||||
# step (migrations, bucket creation). It is done, not broken.
|
||||
IsOneShot = ($f[0] -eq 'exited') -and ($f[3] -eq '0') -and ($f[1] -eq 'none')
|
||||
RoutableByTraefik = ($f[0] -eq 'running') -and ($f[1] -in @('healthy', 'none'))
|
||||
}
|
||||
}
|
||||
|
||||
function Get-StartupWork {
|
||||
param([string]$Name)
|
||||
|
||||
# A container stuck in its entrypoint (apt/dpkg/chown) is starting, not broken.
|
||||
$lines = Invoke-InLxc -Script @"
|
||||
docker top '$Name' -o pid,stat,etime,cmd 2>/dev/null | tail -n +2 || true
|
||||
"@
|
||||
return @($lines | Where-Object { $_ -match '\b(chown|apt|apt-get|dpkg|unzip|tar|cp)\b' })
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "Coolify service readiness - $Uuid" -ForegroundColor Cyan
|
||||
Write-Host ("-" * 72)
|
||||
|
||||
$containers = @(Get-ServiceContainers -ResourceUuid $Uuid)
|
||||
if (-not $containers -or $containers.Count -eq 0) {
|
||||
throw "No container found carrying uuid '$Uuid' in LXC $lxc. Has the service been deployed at all?"
|
||||
}
|
||||
|
||||
$deadline = (Get-Date).AddSeconds($WaitSeconds)
|
||||
$reports = @()
|
||||
|
||||
while ($true) {
|
||||
$reports = @($containers | ForEach-Object { Get-ContainerReport -Name $_ } | Where-Object { $_ })
|
||||
|
||||
$notReady = @($reports | Where-Object { -not $_.RoutableByTraefik -and -not $_.IsOneShot })
|
||||
if ($notReady.Count -eq 0 -or (Get-Date) -ge $deadline) { break }
|
||||
|
||||
$names = ($notReady | ForEach-Object { "$($_.Name)=$($_.Health)" }) -join ', '
|
||||
$left = [int]($deadline - (Get-Date)).TotalSeconds
|
||||
Write-Host " waiting ($left s left): $names" -ForegroundColor DarkGray
|
||||
Start-Sleep -Seconds 10
|
||||
}
|
||||
|
||||
$reports |
|
||||
Select-Object Name, State, Health, FailingStreak,
|
||||
@{ n = 'Routed'; e = { if ($_.IsOneShot) { 'n/a (one-shot)' } else { $_.RoutableByTraefik } } } |
|
||||
Format-Table -AutoSize
|
||||
|
||||
# Healthcheck tuning is the amplifier that turns "slow boot" into "stuck 503".
|
||||
# A short grace window is the amplifier that turns "slow boot" into "stuck 503":
|
||||
# once flagged unhealthy, the container loses its Traefik route entirely.
|
||||
$graceFloorSeconds = 180
|
||||
foreach ($r in $reports) {
|
||||
if ($r.Health -eq 'none' -or $r.HealthStartPeriod) { continue }
|
||||
if (-not $r.HealthInterval -or -not $r.HealthRetries) { continue }
|
||||
|
||||
$grace = $r.HealthInterval * [int]$r.HealthRetries
|
||||
if ($grace -ge $graceFloorSeconds) { continue }
|
||||
|
||||
Write-Host " ! $($r.Name): no start_period; flagged unhealthy after ~$grace s" -ForegroundColor Yellow
|
||||
Write-Host " (interval=$($r.HealthInterval)s x retries=$($r.HealthRetries)). A first boot on this host" -ForegroundColor Yellow
|
||||
Write-Host " can exceed that, and an unhealthy container has no Traefik route -> 503." -ForegroundColor Yellow
|
||||
}
|
||||
|
||||
foreach ($r in $reports) {
|
||||
if ($r.RoutableByTraefik -or $r.IsOneShot) { continue }
|
||||
$work = @(Get-StartupWork -Name $r.Name)
|
||||
if ($work.Count -gt 0) {
|
||||
Write-Host " i $($r.Name) is still running setup work in its entrypoint:" -ForegroundColor DarkCyan
|
||||
$work | ForEach-Object { Write-Host " $_" -ForegroundColor DarkCyan }
|
||||
Write-Host " This is slow-but-progressing, not a failure. Wait, do not redeploy." -ForegroundColor DarkCyan
|
||||
}
|
||||
}
|
||||
|
||||
$target = $Fqdn
|
||||
if (-not $target) {
|
||||
$withFqdn = @($reports | Where-Object { $_.Fqdn })
|
||||
if ($withFqdn.Count -gt 0) { $target = $withFqdn[0].Fqdn }
|
||||
}
|
||||
|
||||
if ($target) {
|
||||
if ($target -notmatch '^https?://') { $target = "https://$target" }
|
||||
Write-Host ""
|
||||
Write-Host "Probing $target" -ForegroundColor Cyan
|
||||
|
||||
$status = $null
|
||||
$body = ''
|
||||
try {
|
||||
$resp = Invoke-WebRequest -Uri $target -UseBasicParsing -TimeoutSec 25
|
||||
$status = [int]$resp.StatusCode
|
||||
$body = [string]$resp.Content
|
||||
}
|
||||
catch {
|
||||
if ($_.Exception.Response) {
|
||||
$status = [int]$_.Exception.Response.StatusCode
|
||||
try {
|
||||
$reader = New-Object IO.StreamReader($_.Exception.Response.GetResponseStream())
|
||||
$body = $reader.ReadToEnd()
|
||||
}
|
||||
catch { $body = '' }
|
||||
}
|
||||
else {
|
||||
Write-Host " transport error: $($_.Exception.Message)" -ForegroundColor Red
|
||||
}
|
||||
}
|
||||
|
||||
if ($status) { Write-Host " HTTP $status" }
|
||||
|
||||
if ($status -eq 503 -and $body -match 'no available server') {
|
||||
Write-Host ""
|
||||
Write-Host " DIAGNOSIS: Traefik has no route for this host." -ForegroundColor Yellow
|
||||
Write-Host " The request fell through to Coolify's catch-all router (priority -1000," -ForegroundColor Yellow
|
||||
Write-Host " service 'noop', empty server list), which is what emits this exact string." -ForegroundColor Yellow
|
||||
Write-Host " Traefik's docker provider only registers containers Docker reports healthy," -ForegroundColor Yellow
|
||||
Write-Host " so an unhealthy/starting container has no route at all." -ForegroundColor Yellow
|
||||
Write-Host " Note: a route that exists but whose backend refuses would return 502, not 503." -ForegroundColor Yellow
|
||||
Write-Host " -> Re-run with -WaitSeconds 600 before changing any configuration." -ForegroundColor Yellow
|
||||
}
|
||||
elseif ($status -ge 200 -and $status -lt 400) {
|
||||
Write-Host " Service is reachable and routed." -ForegroundColor Green
|
||||
}
|
||||
}
|
||||
else {
|
||||
Write-Host " (no FQDN found on the containers; pass -Fqdn to probe)" -ForegroundColor DarkGray
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
$reports
|
||||
+34
-15
@@ -10,26 +10,45 @@ Use this skill to take a project from "local code" to "live on
|
||||
(Proxmox SSH, Coolify API, Cloudflare API) plus new GitHub scaffolding into a
|
||||
single end-to-end pipeline.
|
||||
|
||||
## ⚠️ Instance reality check (READ FIRST — verified 2026-07 on coolify.urieljareth.org, v4.1.2)
|
||||
## ⚠️ Instance reality check (READ FIRST — re-verified 2026-08-29, v4.3.14)
|
||||
|
||||
This instance's REST API is **partial**: the entire `/applications/*` namespace is
|
||||
**404** (create public/dockerfile/private-*, list, get, PATCH, `/envs`, `/logs`).
|
||||
`POST /services` only takes raw compose (no git repo). Consequences:
|
||||
**The `/applications/*` 404 is a Cloudflare edge block, not a Coolify limitation.**
|
||||
Same token, same route: 404 via `https://coolify.urieljareth.org`, **200 via the
|
||||
origin `http://192.168.0.117:8000/api/v1`** (`$env:COOLIFY_API_URL_ORIGIN`) —
|
||||
including `/github-apps`. The instance's own `openapi.yaml` (inside the container
|
||||
at `/var/www/html/openapi.yaml`) declares the full applications namespace, so the
|
||||
complete API contract applies when you call the origin. Two more facts:
|
||||
|
||||
- You **cannot create OR configure a git-based build-from-source app via the API** here.
|
||||
`New-CoolifyApplication.ps1` (POST /applications/public) will 404.
|
||||
- To configure an app that **builds from a (private) git repo** — set its build pack to
|
||||
Docker Compose, its compose location, its **env vars**, and its **per-service domains** —
|
||||
you MUST drive the **web UI with Playwright**. Use `scripts/coolify-ui/*.mjs`
|
||||
(needs `COOLIFY_EMAIL`/`COOLIFY_PASSWORD` in `.env.local.ps1`).
|
||||
- What DOES work via API: `/resources` (inventory+status), `/projects`, `/security/keys`,
|
||||
`/services`, and **`GET /deploy?uuid=&force=true`** (trigger a deploy of any existing app),
|
||||
`GET /deployments/{uuid}` (status+logs).
|
||||
- **State-changing endpoints are POST-only since v4.2** — `GET /deploy?uuid=` now
|
||||
answers 405 `"This endpoint has changed to a POST request."`; use
|
||||
`-Method POST` (see notes §10.1).
|
||||
- The origin is plain HTTP inside the LAN — fine for ops from this machine; do
|
||||
not expose it.
|
||||
|
||||
Consequences:
|
||||
|
||||
- Creating/configuring git-based build-from-source apps **via the API should now
|
||||
work by calling the origin** (`POST /applications/public`, `/dockerfile`,
|
||||
`/private-deploy-key`, …). Not yet exercised end-to-end on 4.3.14 — verify on
|
||||
the next deploy before retiring the UI flow.
|
||||
- The **Playwright UI flow** (`scripts/coolify-ui/*.mjs`, needs
|
||||
`COOLIFY_EMAIL`/`COOLIFY_PASSWORD`) and the **direct DB INSERT** path (§8 of the
|
||||
notes) remain valid fallbacks.
|
||||
- What works through either path: `/resources` (inventory+status), `/projects`,
|
||||
`/security/keys`, `/services`, `/deploy` (POST), `/deployments/{uuid}`.
|
||||
|
||||
Full playbook + gotchas (UTF-8 BOM breaks Coolify's YAML parser, "Reload Compose File" is
|
||||
mandatory, don't queue concurrent deploys, PowerShell `ReadAllText` for key payloads, etc.):
|
||||
**[`references/coolify-4.1.2-notes.md`](references/coolify-4.1.2-notes.md)**. Re-verify the API
|
||||
surface with `GET /version` if the instance was upgraded.
|
||||
**[`references/coolify-4.1.2-notes.md`](references/coolify-4.1.2-notes.md)** (§11 has the
|
||||
2026-08-29 re-verification). Re-verify the API surface with `GET /version` if the instance
|
||||
was upgraded.
|
||||
|
||||
**NEW (2026-07-27):** There is also a **third path** for creating apps — **direct DB
|
||||
INSERT via SSH → pct → docker exec → psql**. See
|
||||
[`references/coolify-4.1.2-notes.md` §8](references/coolify-4.1.2-notes.md).
|
||||
This is the fastest path when you have SSH access to the Proxmox host. The key gotcha: you
|
||||
MUST also INSERT a matching `application_settings` row or deploys crash with
|
||||
"disable_build_cache on null" — and stuck deploy queues must be cleared manually (§8.4).
|
||||
|
||||
## When to use
|
||||
|
||||
|
||||
+64
-15
@@ -6,19 +6,51 @@ All commands assume PowerShell from the project root. Load env first:
|
||||
. .\.env.local.ps1
|
||||
```
|
||||
|
||||
> **Canonical catalog:** [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) — verified
|
||||
> signatures, env requirements, and which scripts are broken on this instance.
|
||||
|
||||
## Which pipelines actually work here (re-verified 2026-08-29, v4.3.14)
|
||||
|
||||
Through the **public hostname**, `/applications/*` still 404s — but that block is
|
||||
**Cloudflare's edge, not Coolify's**: against the origin
|
||||
(`http://192.168.0.117:8000/api/v1`, `$env:COOLIFY_API_URL_ORIGIN`) the full
|
||||
applications namespace responds 200 with the same token. State-changing
|
||||
endpoints are POST-only since v4.2. Script status today:
|
||||
|
||||
| Script | Status | Why |
|
||||
|---|---|---|
|
||||
| `Publish-ProjectToCoolify.ps1` | ⚠️ untested on 4.3.14 | calls `New-CoolifyApplication.ps1` → point `COOLIFY_API_URL` at the origin |
|
||||
| `New-CoolifyApplication.ps1` | ⚠️ untested on 4.3.14 | `POST /applications/public`, `PATCH /applications/{uuid}` — respond via the origin |
|
||||
| `Invoke-CoolifyRollback.ps1` | ⚠️ untested on 4.3.14 | `PATCH /applications/{uuid}` — responds via the origin |
|
||||
| `New-CoolifyService.ps1` | ✅ works | `POST /services` |
|
||||
| `coolify-ui/*.mjs` (Playwright) | ✅ works | drives the web UI |
|
||||
| `New-CoolifyAppViaDB.ps1` | ⚠️ last resort | direct `INSERT` into Coolify's DB |
|
||||
|
||||
**Working paths, in order of preference:**
|
||||
|
||||
1. Multi-container stack with `docker-compose.coolify.yml` → `New-CoolifyService.ps1`.
|
||||
2. Git build-from-source app → try the API against the origin
|
||||
(`New-CoolifyApplication.ps1` with `COOLIFY_API_URL` pointed at the origin);
|
||||
if it misbehaves, fall back to the Playwright UI flow below.
|
||||
3. Last resort → `New-CoolifyAppViaDB.ps1` (no validation, no rollback).
|
||||
|
||||
The scaffold/validate/verify scripts (`Initialize-CoolifyProject.ps1`,
|
||||
`Test-PreDeployChecklist.ps1`, `Test-ServiceOnline.ps1`, `New-GitHubRepo.ps1`)
|
||||
are unaffected and work normally.
|
||||
|
||||
Required env vars (in `.env.local.ps1`, gitignored):
|
||||
|
||||
- `COOLIFY_TOKEN` — for Coolify API.
|
||||
- `GITHUB_TOKEN` — PAT for GitHub API (create repo, list). `git push` uses
|
||||
wincred, NOT this PAT.
|
||||
- (Optional) `GITHUB_OWNER` — override the default `urieljarethbusiness-cpu`.
|
||||
- `COOLIFY_EMAIL` / `COOLIFY_PASSWORD` — Coolify **web UI** login. Required for the
|
||||
UI-driven flow below (this instance's `/applications/*` API is 404 — see
|
||||
`references/coolify-4.1.2-notes.md`).
|
||||
- `COOLIFY_EMAIL` / `COOLIFY_PASSWORD` — Coolify **web UI** login. Required for
|
||||
the UI-driven fallback flow below (see `references/coolify-4.1.2-notes.md`;
|
||||
the probable values are already in `.env.local.ps1`).
|
||||
|
||||
## UI-driven config for git build-from-source apps (this instance, v4.1.2)
|
||||
## UI-driven config for git build-from-source apps (fallback path)
|
||||
|
||||
Because `/applications/*` is 404 here, configure git-based Docker-Compose apps through the
|
||||
If the API-via-origin path misbehaves, configure git-based Docker-Compose apps through the
|
||||
UI with Playwright (run from the manager repo root so `require('playwright')` resolves):
|
||||
|
||||
```powershell
|
||||
@@ -38,8 +70,8 @@ $env:CF_ENVBULK = (Get-Content .\myapp\.env.coolify -Raw) # KEY=VALUE lines ->
|
||||
$env:CF_DOMAINS = '{"web":"https://myapp.urieljareth.org"}' # pin per-service domains (else Coolify assigns random ones -> 503)
|
||||
node .\deploy_skill\scripts\coolify-ui\Configure-CoolifyComposeApp.mjs
|
||||
|
||||
# 3. Trigger the deploy (API works even though app CRUD is 404) and watch it
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=<app>"
|
||||
# 3. Trigger the deploy (POST since v4.2) and watch it
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy?uuid=<app>"
|
||||
# poll GET /deployments/{deployment_uuid} for status+logs; then verify:
|
||||
.\deploy_skill\scripts\Test-ServiceOnline.ps1 -Fqdn https://myapp.urieljareth.org -ExpectTitle "<regex>"
|
||||
```
|
||||
@@ -55,8 +87,16 @@ node .\deploy_skill\scripts\coolify-ui\Configure-CoolifyComposeApp.mjs
|
||||
> Dockerfile builds (POST /applications/public). `New-CoolifyService.ps1` is
|
||||
> for multi-container Docker Compose stacks (POST /services). Use the right
|
||||
> one for your project shape.
|
||||
>
|
||||
> Since 2026-08-29 the first pipeline should also work by pointing
|
||||
> `COOLIFY_API_URL` at the origin (`http://192.168.0.117:8000/api/v1`) — the 404
|
||||
> on `/applications/*` was Cloudflare's edge, not Coolify's. Verify on the next
|
||||
> deploy before relying on it.
|
||||
|
||||
### Full pipeline (one shot)
|
||||
### Full pipeline (one shot) — via the origin
|
||||
|
||||
Blocked only through the public hostname (Cloudflare 404 on
|
||||
`/applications/*`). Point the API at the origin and step 5 should succeed:
|
||||
|
||||
```powershell
|
||||
.\deploy_skill\scripts\Publish-ProjectToCoolify.ps1 `
|
||||
@@ -168,18 +208,27 @@ You will be prompted for the SHA if you omit `-CommitSha`.
|
||||
|
||||
## Coolify API (already documented in coolify_skill)
|
||||
|
||||
The deploy skill reuses `coolify_skill/scripts/Invoke-CoolifyApi.ps1`. Useful
|
||||
endpoints for the deploy flow:
|
||||
The deploy skill reuses `coolify_skill/scripts/Invoke-CoolifyApi.ps1`. Endpoints
|
||||
that **work here**, useful for the deploy flow:
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/projects"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/servers"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications/<uuid>"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/applications/<uuid>/deployments"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy" -BodyJson (@{ uuid = "<uuid>" } | ConvertTo-Json)
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services"
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" # enumerate apps + get uuids
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method POST -Path "/deploy?uuid=<uuid>" # trigger a deploy (POST-only since v4.2)
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments/<deployment-uuid>"
|
||||
```
|
||||
|
||||
⚠️ `/applications*` and `/github-apps` **404 through the public hostname
|
||||
(Cloudflare edge block, re-verified 2026-08-29)** — call them via the origin
|
||||
(`$env:COOLIFY_API_URL_ORIGIN` = `http://192.168.0.117:8000/api/v1`), where the
|
||||
full namespace works. `/resources` remains the simplest inventory;
|
||||
`/deployments/<uuid>` covers deployment status.
|
||||
|
||||
Remember `-Raw` if you need objects instead of a JSON string — see
|
||||
[`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) §1.1.
|
||||
|
||||
For per-endpoint schemas, search `coolify_skill/references/ops/` with `rg`.
|
||||
|
||||
## Templates
|
||||
@@ -239,7 +288,7 @@ the manager repo root.
|
||||
| Symptom | Likely cause | Fix |
|
||||
|---------|--------------|-----|
|
||||
| `git push` asks for username/password | wincred cache miss | Run any git HTTPS op once interactively; or `cmdkey /generic:git:https://github.com /user:urieljarethbusiness-cpu /pass:<PAT>` |
|
||||
| GitHub API 401 | `GITHUB_TOKEN` not loaded or expired | `. .\.env.local.ps1`; regenerate PAT at https://github.com/settings/tokens |
|
||||
| GitHub API 401 | `GITHUB_TOKEN` not loaded or expired | `. .\.env.local.ps1` (carries the wincred token since 2026-08-29); re-extract with `git credential fill` if it rotates |
|
||||
| Coolify API 401 | `COOLIFY_TOKEN` not loaded | `. .\.env.local.ps1` |
|
||||
| App at 502 after deploy | `ports_exposes` mismatch (default is 3000) | `New-CoolifyApplication.ps1 -PortsExposes <real-port>` |
|
||||
| App can't reach DB | used `localhost` or wrong network | Use service name + external `coolify` network (see §2.1, §2.2 in AGENTS-coolify-apps.md) |
|
||||
|
||||
@@ -7,6 +7,12 @@
|
||||
|
||||
## 1. The REST API surface is PARTIAL — `/applications/*` is 404
|
||||
|
||||
> **CORRECCIÓN (2026-08-29, §11):** ese 404 lo imponía **Cloudflare en el
|
||||
> hostname público, no Coolify**. Contra el origen
|
||||
> (`http://192.168.0.117:8000/api/v1`) el namespace completo responde 200 y el
|
||||
> `openapi.yaml` del contenedor declara toda la superficie. Esta sección y la
|
||||
> tabla se conservan como registro histórico del diagnóstico de 2026-08-07.
|
||||
|
||||
With a valid **root-team** token (`GET /teams/current` → "Root Team"):
|
||||
|
||||
| Endpoint | Result |
|
||||
@@ -114,7 +120,7 @@ Realtime, Storage, Kong, Studio) end-to-end. New, reusable facts:
|
||||
|
||||
### 7.2 Locally-built images → Coolify's deploy `pull`s and fails
|
||||
- Coolify's service deploy runs `docker compose pull` → `pull access denied ... repository does not exist` for a local-only image tag. **Don't use Coolify's Deploy button/`/start` for local images.**
|
||||
- Instead build the image on the server (clone repo + `docker build -t <tag>`), then `docker compose up -d` **manually** in `/data/coolify/services/<uuid>/` (default pull policy skips pull when the image exists locally). Same pattern as `Deploy-SoloLeveling.ps1`.
|
||||
- Instead build the image on the server (clone repo + `docker build -t <tag>`), then `docker compose up -d` **manually** in `/data/coolify/services/<uuid>/` (default pull policy skips pull when the image exists locally). Same pattern as `scripts/apps/Deploy-SoloLeveling.ps1`.
|
||||
- Coolify's normalized on-disk compose (from your base64 raw) **preserves** your `image`, inlined `environment`, custom `labels` (incl. Traefik) and networks — but **renames containers to `<service>-<uuid>`**. Cross-container refs must use the *other* service's real name (unchanged), not the renamed one.
|
||||
- The service dir + its per-service external network `<uuid>` are created only on Coolify's own deploy. For a first manual `up`, run `docker network create --attachable <uuid>` first (else `network <uuid> declared as external, but could not be found`). Put app containers on the external `coolify` network to reach other stacks (Traefik `coolify-proxy` is already on it).
|
||||
|
||||
@@ -141,7 +147,7 @@ Every deploy script here sets `$ErrorActionPreference = "Stop"` and shells out t
|
||||
kills the script on its first line of benign progress (e.g. git `Cloning into 'repo'...`).
|
||||
Symptom: exit 1 with the error anchored at `& ssh @sshArgs` in `ProxmoxAgent.ps1`, right after the
|
||||
first `git`/`ssh` progress line — **before any real work fails**.
|
||||
- **Fix:** invoke the script **plain** — `& .\Deploy-SoloLeveling.ps1` (no `2>&1`/`*>&1`, no
|
||||
- **Fix:** invoke the script **plain** — `& .\scripts\apps\Deploy-SoloLeveling.ps1` (no `2>&1`/`*>&1`, no
|
||||
merging pipe). The harness/terminal already captures the process's stderr at the OS level, which
|
||||
does **not** create ErrorRecords. Same rule for `Publish-ProjectToCoolify.ps1`,
|
||||
`New-CoolifyService.ps1`, `Invoke-CoolifyRollback.ps1`, `Test-ServiceOnline.ps1`.
|
||||
@@ -150,7 +156,7 @@ first `git`/`ssh` progress line — **before any real work fails**.
|
||||
- Note the scripts' *internal* `git push 2>&1 | ForEach-Object {…}` is fine (scoped to one
|
||||
statement); the trap is the **outer** redirect the caller adds.
|
||||
|
||||
### 7.8 build-on-server re-deploy: verification + rollback caveats (Solo Leveling, `Deploy-SoloLeveling.ps1`)
|
||||
### 7.8 build-on-server re-deploy: verification + rollback caveats (Solo Leveling, `scripts/apps/Deploy-SoloLeveling.ps1`)
|
||||
- **Confirm the shipped commit, not just health.** The build step echoes `HEAD: <sha> <subject>`
|
||||
from the fresh clone — gate on it matching your intended commit, and confirm the app container
|
||||
shows **`Recreated`** (not reused) in `docker compose up -d` output. Health 200 alone only proves
|
||||
@@ -162,3 +168,504 @@ first `git`/`ssh` progress line — **before any real work fails**.
|
||||
If instant rollback matters, tag per-commit too (`:<sha>` alongside `:latest`) so you can retag
|
||||
`:latest` to a prior digest and `docker compose up -d`. (The skill's `Invoke-CoolifyRollback.ps1`
|
||||
assumes the git-build path, which is `/applications/*` = 404 here — it doesn't apply to build-on-server.)
|
||||
|
||||
## 8. Creating git-based apps via direct DB INSERT (when UI credentials are unavailable)
|
||||
|
||||
> Verified 2026-07-27 deploying `AgendaMax` (Node 22 + Vite + Express + SQLite,
|
||||
> Dockerfile build pack, GitHub source). This is the **third** path, alongside
|
||||
> §3 (Playwright UI) and §6 (API, 404 here). Use it when you have SSH+DB access
|
||||
> but NOT the Coolify web UI email/password.
|
||||
|
||||
### 8.1 The SSH → pct → docker exec → psql chain
|
||||
|
||||
```
|
||||
local machine
|
||||
→ ssh [email protected] -i keys\proxmox_ed25519
|
||||
→ pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -c "SQL_HERE"
|
||||
```
|
||||
|
||||
**Credentials (all local/private, no secrets leave the host):**
|
||||
- Proxmox host: `[email protected]`
|
||||
- SSH key: `keys\proxmox_ed25519`
|
||||
- Coolify LXC: `102`
|
||||
- Coolify DB container: `coolify-db` (PostgreSQL 15, user `coolify`, db `coolify`, no password — container-internal)
|
||||
- Coolify app container: `coolify` (image `ghcr.io/coollabsio/coolify:4.1.2`)
|
||||
- Coolify API token: in `.env.local.ps1` → `$env:COOLIFY_TOKEN`
|
||||
- GitHub PAT: in `.env.local.ps1` → `$env:GITHUB_TOKEN` (scope `repo`, account `urieljarethbusiness-cpu`)
|
||||
|
||||
### 8.2 PowerShell quoting nightmare — the scp+sh workaround
|
||||
|
||||
**Problem:** Passing SQL (with single quotes, backslashes in PHP namespaces like
|
||||
`App\Models\GithubApp`, or `{{pr_id}}` braces) through the chain
|
||||
PowerShell → ssh → pct → docker → psql is quoting hell. Nested `'…'` inside
|
||||
`"…"` inside `"…"` breaks at every level.
|
||||
|
||||
**Solution:** Write a `.sh` script locally → `scp` to the Proxmox host → `ssh … sh /tmp/script.sh`.
|
||||
|
||||
```powershell
|
||||
# 1. Write the script locally with heredoc-safe content
|
||||
# 2. scp it to the host
|
||||
scp -i $SSH_KEY script.sh root@192.168.0.200:/tmp/script.sh
|
||||
# 3. Execute remotely
|
||||
ssh -i $SSH_KEY root@192.168.0.200 "sh /tmp/script.sh"
|
||||
```
|
||||
|
||||
**Never** try to pass complex SQL inline through `ssh … "pct exec … psql … -c '…'"` from
|
||||
PowerShell — the nested quoting will eat hours. Always scp a script.
|
||||
|
||||
### 8.3 Step-by-step: create a Dockerfile-based app from a GitHub repo
|
||||
|
||||
**Prerequisites:**
|
||||
- The repo must be on GitHub (`urieljarethbusiness-cpu/<name>`) — Coolify's GitHub App
|
||||
(source_id=3, installation_id=121211999) is already configured and can access all repos
|
||||
under that account.
|
||||
- Push the code first: `git push github main`.
|
||||
- Get the GitHub repo numeric ID: `GET https://api.github.com/repos/<owner>/<repo>` → `.id`.
|
||||
|
||||
**Step 1 — Find the environment_id:**
|
||||
```sql
|
||||
SELECT id FROM environments WHERE uuid = '<env_uuid>';
|
||||
-- e.g. for project "tools" production: returns 9
|
||||
```
|
||||
|
||||
**Step 2 — INSERT into `applications`:** every field with a NOT NULL DEFAULT in the schema
|
||||
must be set explicitly (the INSERT doesn't fire Laravel's model events, so defaults from
|
||||
migrations are the DB column defaults, not Eloquent `$casts`/boot logic).
|
||||
|
||||
Key fields for a Dockerfile app:
|
||||
```sql
|
||||
INSERT INTO applications (
|
||||
uuid, name, git_repository, git_branch, git_commit_sha,
|
||||
build_pack, -- 'dockerfile'
|
||||
dockerfile_location, -- '/Dockerfile'
|
||||
ports_exposes, -- '3000' (your app's port)
|
||||
health_check_path, health_check_port, health_check_host,
|
||||
health_check_method, health_check_return_code, health_check_scheme,
|
||||
health_check_interval, health_check_timeout, health_check_retries, health_check_start_period,
|
||||
health_check_enabled, -- false (simpler; enable later via UI if needed)
|
||||
limits_memory, limits_memory_swap, limits_memory_swappiness, limits_memory_reservation,
|
||||
limits_cpus, limits_cpu_shares,
|
||||
status, -- 'exited'
|
||||
preview_url_template, -- '{{pr_id}}.{{domain}}'
|
||||
fqdn, -- 'https://<name>.urieljareth.org'
|
||||
repository_project_id, -- numeric GitHub repo ID (from GitHub API)
|
||||
source_type, -- 'App\Models\GithubApp'
|
||||
source_id, -- 3 (the GitHub App, see SELECT id FROM github_apps)
|
||||
destination_type, -- 'App\Models\StandaloneDocker'
|
||||
destination_id, -- 0 (the localhost Docker, see SELECT id FROM standalone_dockers)
|
||||
environment_id, -- from Step 1
|
||||
base_directory, -- '/'
|
||||
static_image, -- 'nginx:alpine' (unused for dockerfile pack, but NOT NULL)
|
||||
created_at, updated_at -- NOW()
|
||||
) VALUES (...);
|
||||
```
|
||||
|
||||
**Step 3 — INSERT into `application_settings` (CRITICAL — skip this and deploys crash):**
|
||||
```
|
||||
production.ERROR: Attempt to read property "disable_build_cache" on null
|
||||
at ApplicationDeploymentJob.php:214
|
||||
```
|
||||
Coolify's deployment job reads `$application->settings->disable_build_cache` in its
|
||||
constructor. Without a settings row, `settings` is null and the job dies instantly —
|
||||
the deployment stays "in_progress" forever and blocks the queue (see §8.4).
|
||||
|
||||
```sql
|
||||
INSERT INTO application_settings (
|
||||
application_id, -- the id from Step 2's RETURNING
|
||||
is_static, is_git_submodules_enabled, is_git_lfs_enabled,
|
||||
is_auto_deploy_enabled, is_force_https_enabled, is_debug_enabled,
|
||||
is_preview_deployments_enabled, is_log_drain_enabled, is_gpu_enabled,
|
||||
is_swarm_only_worker_nodes, is_raw_compose_deployment_enabled,
|
||||
is_build_server_enabled, is_consistent_container_name_enabled,
|
||||
is_gzip_enabled, is_stripprefix_enabled,
|
||||
is_container_label_escape_enabled, is_container_label_readonly_enabled,
|
||||
disable_build_cache, is_spa, is_git_shallow_clone_enabled,
|
||||
is_pr_deployments_public_enabled, use_build_secrets, inject_build_args_to_dockerfile,
|
||||
docker_images_to_keep,
|
||||
created_at, updated_at
|
||||
) VALUES (
|
||||
<app_id>, false, true, true, true, true, false, false, false, false,
|
||||
true, false, false, false, true, true, true, true, false, false, true,
|
||||
false, false, true, 2, NOW(), NOW()
|
||||
);
|
||||
```
|
||||
|
||||
**Step 4 — Trigger deploy via API (works even though app CRUD is 404):**
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
$h = @{ Authorization = "Bearer $env:COOLIFY_TOKEN" }
|
||||
Invoke-RestMethod "$env:COOLIFY_API_URL/deploy?uuid=<app_uuid>&force=true" -Headers $h
|
||||
# Returns: { deployments: [{ deployment_uuid: "..." }] }
|
||||
```
|
||||
|
||||
Monitor: `GET /deployments/<deployment_uuid>` → status goes `queued → in_progress → finished`.
|
||||
|
||||
### 8.4 Stuck deployment queue — how to unblock
|
||||
|
||||
When a deployment fails (e.g. missing `application_settings`), the queue row stays
|
||||
`in_progress` forever and blocks ALL subsequent deploys of that app.
|
||||
|
||||
**Diagnose:**
|
||||
```sql
|
||||
SELECT id, status, created_at FROM application_deployment_queues
|
||||
WHERE application_id = '<app_id>' ORDER BY id DESC LIMIT 5;
|
||||
```
|
||||
|
||||
**Fix:**
|
||||
```sql
|
||||
-- Mark the stuck job as failed
|
||||
UPDATE application_deployment_queues SET status = 'failed', updated_at = NOW()
|
||||
WHERE id = <stuck_id>;
|
||||
-- Delete ALL queue entries for the app and start clean
|
||||
DELETE FROM application_deployment_queues WHERE application_id = '<app_id>';
|
||||
```
|
||||
|
||||
Then restart the Coolify queue worker (so it picks up new jobs):
|
||||
```sh
|
||||
pct exec 102 -- docker exec coolify php artisan queue:restart
|
||||
```
|
||||
|
||||
Wait 5 seconds, then trigger a fresh deploy via the API.
|
||||
|
||||
### 8.5 Reference: existing GitHub-source apps on this instance (verified 2026-07-27)
|
||||
|
||||
| id | name | repo | build_pack | source_id |
|
||||
|----|------|------|------------|-----------|
|
||||
| 43 | baserow | baserow/baserow | dockercompose | 3 |
|
||||
| 46 | estación-de-documentos | urieljarethbusiness-cpu/Estación-de-Documentos | dockercompose | 3 |
|
||||
| 47 | cotizador | urieljarethbusiness-cpu/cotizador | dockercompose | 3 |
|
||||
| 48 | firecrawl | firecrawl/firecrawl | dockercompose | 3 |
|
||||
| 50 | audio-a-texto | urieljarethbusiness-cpu/Audio-a-Texto | dockerfile | 3 |
|
||||
| 51 | agendamax | urieljarethbusiness-cpu/agendamax | dockerfile | 3 |
|
||||
|
||||
GitHub App: id=3, uuid=`miakw8c0vthrtweroh9kzy1t`, app_id=3266541, installation_id=121211999.
|
||||
Public GitHub source: id=0, uuid=`yyq0od5j3xkdf8twc28n6coh`.
|
||||
Standalone Docker (destination): id=0, uuid=`jxlxd62k0d8owl6orgkjj1ob`, network=`coolify`.
|
||||
Server: id=0, uuid=`l10mdaago0z605pga93gl6cz`, name=`localhost`.
|
||||
|
||||
### 8.6 Gitea repos as build source — not directly supported via this path
|
||||
|
||||
Coolify's GitHub App source only works with GitHub.com repos. For a Gitea-hosted repo:
|
||||
- **Option A (recommended):** mirror to GitHub (`git remote add github …; git push github main`),
|
||||
then create the Coolify app from the GitHub repo. This is what we did for AgendaMax —
|
||||
the primary repo stays on Gitea, GitHub is just a deploy mirror.
|
||||
- **Option B:** create a "Private repository (deploy key)" app via the web UI — set
|
||||
`git_repository` to the Gitea SSH URL and `private_key_id` to an SSH key with Gitea access.
|
||||
Gitea SSH is on port `22222` (mapped from container 22). This path needs UI credentials
|
||||
and hasn't been tested on this instance.
|
||||
|
||||
Re-verified 2026-07-31 (`prompt-gallery-e3`): Option A still works, and the GitHub App
|
||||
(source_id=3) reaches a **freshly created** private repo under `urieljarethbusiness-cpu`
|
||||
with no extra configuration — the installation covers all repos on the account, so there
|
||||
is nothing to click between `POST /user/repos` and the first deploy.
|
||||
|
||||
### 8.7 Env vars are Laravel-`encrypted` — raw SQL INSERT kills the deploy
|
||||
|
||||
> Verified 2026-07-31 deploying `prompt-gallery-e3` (Next.js 16 + `node:sqlite`).
|
||||
> §8.3 creates the app but says nothing about env vars; this is the trap.
|
||||
|
||||
`environment_variables.value` carries Laravel's `encrypted` cast. Insert a **plaintext**
|
||||
value with `psql` and the app row looks perfect, but every deploy dies **after** cloning
|
||||
and reading the Dockerfile, with no hint about which field is at fault:
|
||||
|
||||
```
|
||||
Deployment failed: The payload is invalid.
|
||||
Error type: Illuminate\Contracts\Encryption\DecryptException
|
||||
Location: /var/www/html/vendor/laravel/framework/src/Illuminate/Encryption/Encrypter.php:244
|
||||
```
|
||||
|
||||
A correctly stored value is base64 JSON (`eyJpdiI6…`, `{iv,value,mac,tag}`), 200–300 chars
|
||||
for short secrets. The `APP_KEY` never leaves the container, so **don't try to encrypt from
|
||||
the outside** — create the rows through Eloquent instead:
|
||||
|
||||
```powershell
|
||||
# write PHP locally -> scp to host -> pct push -> docker cp -> tinker
|
||||
pct push 102 /tmp/envs.php /tmp/envs.php
|
||||
pct exec 102 -- docker cp /tmp/envs.php coolify:/tmp/envs.php
|
||||
pct exec 102 -- docker exec coolify php artisan tinker --execute='include "/tmp/envs.php";'
|
||||
```
|
||||
|
||||
```php
|
||||
$fila = new \App\Models\EnvironmentVariable();
|
||||
$fila->key = 'SESSION_SECRET';
|
||||
$fila->value = '…'; // se cifra al guardar
|
||||
$fila->resourceable_type = \App\Models\Application::class;
|
||||
$fila->resourceable_id = 53;
|
||||
$fila->save(); // uuid se autogenera
|
||||
```
|
||||
|
||||
Three gotchas found doing this:
|
||||
|
||||
- **DELETE then CREATE — never UPDATE a plaintext row.** Saving over an existing bad row
|
||||
throws the same `DecryptException` first, because a model hook reads `value` before your
|
||||
assignment is written. Wipe the rows with SQL, then create them via Eloquent.
|
||||
- **The column is `is_buildtime`, not `is_build_time`** (and `is_runtime`/`is_buildtime`
|
||||
are set by the model — don't touch them).
|
||||
- **Two rows per key is correct.** Coolify mirrors every variable into a preview copy
|
||||
(`is_preview = true`), so ids come in pairs. The pre-existing apps show the same shape;
|
||||
it is not a duplicate-insert bug.
|
||||
|
||||
### 8.8 Persistent volumes by this path
|
||||
|
||||
`local_persistent_volumes` uses `resource_type` / `resource_id` — **not** the
|
||||
`resourceable_*` names that `environment_variables` uses. Plain SQL is fine here (no
|
||||
encrypted columns):
|
||||
|
||||
```sql
|
||||
INSERT INTO local_persistent_volumes (name, mount_path, resource_type, resource_id, uuid, created_at, updated_at)
|
||||
VALUES ('<app_uuid>-galeria-datos', '/app/data', 'App\Models\Application', <app_id>, '<uuid24>', NOW(), NOW());
|
||||
```
|
||||
|
||||
Insert the volume **before** the first deploy. `New-CoolifyAppViaDB.ps1` triggers a deploy
|
||||
as its last step, so for an app that needs env vars or a volume, do the whole set of rows
|
||||
in one data-modifying-CTE statement and trigger `GET /deploy?uuid=` yourself — otherwise
|
||||
the first build is guaranteed to fail and you burn ~15 min of Next.js build time.
|
||||
|
||||
### 8.9 Script bugs fixed 2026-07-31 (were silently breaking repo creation)
|
||||
|
||||
- **`" 20[04-9] "` accepted 200 and 204–209 but rejected 201/202/203** — a character-class
|
||||
typo for `[0-9]`. Present in **both** `gitea_skill/scripts/Invoke-GiteaApi.ps1` and
|
||||
`deploy_skill/scripts/Invoke-GitHubApi.ps1`. Every repo/hook/release creation returns
|
||||
**201**, so it threw *after* successfully creating the resource — leaving a created repo
|
||||
and an aborted pipeline. Both fixed to `" 20[0-9] "`.
|
||||
- **`Invoke-GitHubApi.ps1` BOM bug (documented in §7.3) is now actually fixed** — the POST
|
||||
body is written with `[IO.File]::WriteAllText(..., UTF8Encoding($false))`.
|
||||
- **`git push` to GitHub: `Authorization: Bearer <PAT>` does NOT work.** git-over-https
|
||||
wants Basic. For a headless push that leaves no token on disk:
|
||||
|
||||
```powershell
|
||||
$basic = [Convert]::ToBase64String([Text.Encoding]::ASCII.GetBytes("x-access-token:$env:GITHUB_TOKEN"))
|
||||
git -C $repo -c "http.extraHeader=Authorization: Basic $basic" -c "credential.helper=" push -u github main
|
||||
```
|
||||
|
||||
(`Sync-GiteaRemote.ps1`'s `token <T>` header is right for Gitea; it is not for GitHub.)
|
||||
- `New-GitHubRepo.ps1` forces `auto_init = $true`, which puts a commit on the remote and
|
||||
makes pushing existing history a non-fast-forward. For a mirror of a repo that already
|
||||
has history, POST `/user/repos` yourself with `auto_init = $false`.
|
||||
|
||||
## 9. Apps with a sibling database (verified 2026-07-31, `demospa` — Next.js 15 + Prisma + MySQL 8)
|
||||
|
||||
Deployed `serenidad-spa` as `demospa.urieljareth.org` (app id 54) with a Coolify-managed
|
||||
MySQL. New, reusable facts beyond §8:
|
||||
|
||||
### 9.1 `POST /databases/mysql` WORKS — the database API is not part of the 404 namespace
|
||||
|
||||
Unlike `/applications/*`, the `/databases/*` namespace responds. `POST /databases/mysql`
|
||||
with `{server_uuid, project_uuid, environment_name, environment_uuid, destination_uuid,
|
||||
name, image, mysql_root_password, mysql_database, mysql_user, mysql_password}` returns
|
||||
`{uuid, internal_db_url}`. **The DB's internal hostname IS its uuid** — so
|
||||
`DATABASE_URL=mysql://user:pass@<db-uuid>:3306/<db>`. Both the DB and dockerfile-pack apps
|
||||
land on the external `coolify` network, so service-name DNS works with no extra wiring.
|
||||
|
||||
Two gotchas:
|
||||
- **`instant_deploy: true` did NOT start it.** The resource was created with
|
||||
`status=exited:unhealthy` and no container. `GET /databases/{uuid}/start` then answered
|
||||
`400 {"message":"Database is already running."}` (the status field lies). What actually
|
||||
started it: **`GET /deploy?uuid=<db-uuid>`** — the same trigger used for apps.
|
||||
- **Never call `/start` and `/deploy` back to back.** Doing so recreated the container
|
||||
mid-initialization and left a partial datadir; MySQL then crash-looped forever on
|
||||
`--initialize specified but the data directory has files in it`. Recovery =
|
||||
`docker compose down` in `/data/coolify/databases/<uuid>/`, `docker volume rm
|
||||
mysql-data-<uuid>`, then one clean `up -d`. Same "don't queue deploys" rule as §4.
|
||||
|
||||
### 9.2 MySQL 8 init on this host takes ~7 min, and `start_period` is hardcoded to 5s
|
||||
|
||||
Coolify's generated DB compose sets `healthcheck.start_period: 5s`, but a first-time
|
||||
MySQL 8 init here takes minutes (InnoDB init alone ~53s). Patch
|
||||
`/data/coolify/databases/<uuid>/docker-compose.yml` to a longer `start_period` before the
|
||||
first `up -d`.
|
||||
|
||||
**Do not trust `mysqladmin ping` as a readiness signal.** The MySQL entrypoint runs a
|
||||
*temporary* server with `--skip-networking` while it creates the database and user, so the
|
||||
socket answers (and Coolify reports `healthy`) while TCP still refuses connections. The log
|
||||
even prints `ready for connections ... port: 0` for that temp server. Gate on TCP:
|
||||
`mysqladmin -h127.0.0.1 -uroot -p"$MYSQL_ROOT_PASSWORD" ping`, and confirm a
|
||||
`ready for connections ... port: 3306` line.
|
||||
|
||||
### 9.3 Prisma + MySQL 8: pin `mysql_native_password`
|
||||
|
||||
Add to the DB service's compose `command:`
|
||||
`--default-authentication-plugin=mysql_native_password --character-set-server=utf8mb4
|
||||
--collation-server=utf8mb4_unicode_ci`. 8.0.46 only warns that the flag is deprecated. Set
|
||||
it **before** the first init so the created user gets native auth (afterwards the user
|
||||
already exists and the env vars are ignored — you'd need `ALTER USER`).
|
||||
|
||||
### 9.4 The rolling-update healthcheck window is ~2 minutes — do slow work in the background
|
||||
|
||||
Coolify polls the container's **Dockerfile `HEALTHCHECK`** ~6 times at 30s and aborts with
|
||||
"New container is not healthy, rolling back" if it hasn't passed. An entrypoint that runs
|
||||
`prisma db push` + seed *before* starting the server will lose this race on a cold DB — the
|
||||
build succeeds and the deploy still fails.
|
||||
|
||||
Working shape: start the server in the foreground and run DB preparation in a background
|
||||
subshell, so the health endpoint answers in seconds while data lands moments later.
|
||||
|
||||
```sh
|
||||
preparar_base_de_datos() { ...db push retry loop...; ...seed...; }
|
||||
preparar_base_de_datos &
|
||||
exec "$@" # el proceso en segundo plano sobrevive al exec
|
||||
```
|
||||
|
||||
Keep the Dockerfile healthcheck tight (`--start-period=10s --interval=10s`) so it turns
|
||||
healthy inside Coolify's window, and give the health endpoint **no DB dependency**.
|
||||
|
||||
### 9.5 `inject_build_args_to_dockerfile` bakes every env var into the image
|
||||
|
||||
`New-CoolifyAppViaDB.ps1`'s settings row sets this `true`, so Coolify rewrites the
|
||||
Dockerfile with an `ARG`/`ENV` per variable — the build log fills with
|
||||
`SecretsUsedInArgOrEnv: ... (ARG "AUTH_SECRET")` and the secrets end up in image layers,
|
||||
violating §2.5 of `AGENTS-coolify-apps.md`. Set it to `false` unless the app genuinely needs
|
||||
build-time vars (a Next.js app only does if it reads `NEXT_PUBLIC_*`, which are inlined at
|
||||
build time — grep the source before deciding).
|
||||
|
||||
### 9.6 Pushing to the GitHub mirror auto-triggers a deploy
|
||||
|
||||
The settings row also sets `is_auto_deploy_enabled = true`, and Coolify's GitHub App gets
|
||||
push webhooks for the whole account — so `git push github main` queues a deploy on its own.
|
||||
Expect an extra `in_progress` row in `application_deployment_queues`; clear it (§8.4) before
|
||||
triggering your own, or just let the automatic one run.
|
||||
|
||||
### 9.7 Next.js: pages that query the DB break `docker build`
|
||||
|
||||
Any App Router page that hits the database without `export const dynamic = "force-dynamic"`
|
||||
is prerendered during `next build`, where no DB exists. Pages reading `cookies()` are
|
||||
already dynamic; public landing/catalog pages usually are not. Validate locally with a
|
||||
deliberately unreachable `DATABASE_URL` — the build must still finish (Prisma logs errors
|
||||
but they are non-fatal once every DB page is `ƒ`).
|
||||
|
||||
`sharp` needs no special handling: `npm ci` on `node:22-alpine` resolves
|
||||
`@img/sharp-linuxmusl-x64` as long as the lockfile was generated with all platform variants
|
||||
(it is, by default), so remote-image optimization works.
|
||||
|
||||
---
|
||||
|
||||
## 10. Correcciones verificadas 2026-08-27 (despliegue de `escudoverde-site`)
|
||||
|
||||
### 10.1 `/deploy` ahora exige POST, no GET
|
||||
|
||||
El §8.4 dice `GET /deploy?uuid=&force=true`. **Hoy responde 405 Method Not Allowed.**
|
||||
Con `-Method POST` funciona y devuelve el `deployment_uuid` normalmente:
|
||||
|
||||
```powershell
|
||||
Invoke-RestMethod "$env:COOLIFY_API_URL/deploy?uuid=<app_uuid>&force=true" -Headers $h -Method POST
|
||||
```
|
||||
|
||||
`GET /deployments/{uuid}` sigue funcionando igual para el seguimiento.
|
||||
|
||||
### 10.2 El `GITHUB_TOKEN` de `.env.local.ps1` está caducado
|
||||
|
||||
`New-GitHubRepo.ps1` falla con `401 Bad credentials`. El token que **sí** sirve es el que
|
||||
guarda el Administrador de credenciales de Windows para `github.com` (cuenta
|
||||
`urieljarethbusiness-cpu`, scopes `gist, repo, workflow`). Se recupera sin exponerlo:
|
||||
|
||||
```bash
|
||||
TOK=$(printf "protocol=https\nhost=github.com\n\n" | git credential fill | sed -n 's/^password=//p')
|
||||
```
|
||||
|
||||
Conviene rotar el de `.env.local.ps1` o hacer que los scripts caigan a `git credential fill`.
|
||||
|
||||
### 10.3 Imágenes nginx sin root: el orden dentro del `RUN` importa
|
||||
|
||||
El builder de Coolify corre el `docker build` **sin DAC override para root**. Consecuencia
|
||||
concreta con la receta habitual de nginx no-root:
|
||||
|
||||
- Si `nginx -t` va **después** del `chown` de `/tmp` al usuario `nginx`, el build falla con
|
||||
`open() "/tmp/nginx.pid" failed (13: Permission denied)` — aunque en local funcione.
|
||||
- `nginx -t` **crea** el fichero pid. Si se queda en la imagen, pertenece a root y el
|
||||
contenedor no arranca al hacer `USER nginx`.
|
||||
|
||||
Orden que funciona en ambos lados:
|
||||
|
||||
```dockerfile
|
||||
RUN set -eux; \
|
||||
mkdir -p /tmp/nginx/client_body /tmp/nginx/proxy; \
|
||||
nginx -t -c /etc/nginx/nginx.conf; \
|
||||
rm -f /tmp/nginx/nginx.pid; \
|
||||
chown -R nginx:nginx /tmp/nginx /usr/share/nginx/html /var/cache/nginx; \
|
||||
chmod 0777 /tmp/nginx
|
||||
USER nginx
|
||||
```
|
||||
|
||||
### 10.4 `immutable` + nombre de fichero fijo = despliegue invisible
|
||||
|
||||
Con `Cache-Control: public, max-age=31536000, immutable` sobre `/assets/`, Cloudflare
|
||||
sirvió el CSS anterior durante todo el despliegue siguiente (`cf-cache-status: HIT`,
|
||||
`Age: 1008`). El contenedor tenía la versión nueva; el usuario veía la vieja.
|
||||
|
||||
**Regla para cualquier app estática en esta instancia:** o los assets llevan hash en el
|
||||
nombre, o el HTML los referencia con `?v=<hash del contenido>`. Sin eso, `immutable` es
|
||||
una trampa: el despliegue "funciona" y no cambia nada visible.
|
||||
|
||||
---
|
||||
|
||||
## 11. Re-verificación completa 2026-08-29 (v4.3.14): el 404 de `/applications` era Cloudflare
|
||||
|
||||
Auditoría integral con SSH + APIs validadas. La instancia corre ahora
|
||||
**`4.3.14`** (`GET /version`), actualizada desde 4.3.10. Hallazgos, en orden de
|
||||
importancia:
|
||||
|
||||
### 11.1 La API está COMPLETA — el 404 de `/applications/*` lo impone Cloudflare, no Coolify
|
||||
|
||||
Mismo token de siempre, misma ruta, dos caminos:
|
||||
|
||||
| Llamada | Resultado |
|
||||
|---|---|
|
||||
| `https://coolify.urieljareth.org/api/v1/applications` (público, vía Cloudflare) | **404** |
|
||||
| `http://192.168.0.117:8000/api/v1/applications` (origen, LXC 102) | **200** (JSON completo) |
|
||||
| `/github-apps` | 404 vía CF · **200 vía origen** |
|
||||
| `/dockerfiles`, `/sources`, `/notifications`, `/private-keys` | 404 en ambos → **no existen como rutas**; los nombres correctos están en el spec (ver 11.2) |
|
||||
| `/version`, `/resources`, `/servers`, `/projects`, `/teams`, `/services`, `/databases`, `/deployments`, `/security/keys` | 200 por ambas vías |
|
||||
|
||||
Todo el diagnóstico de §1 ("API parcial", v4.1.2) era este mismo bloqueo de
|
||||
Cloudflare, no un recorte de la API. Las secciones §3 (Playwright) y §8 (DB
|
||||
INSERT) siguen siendo fallbacks válidos, pero **la vía API debería funcionar
|
||||
llamando al origen** (`$env:COOLIFY_API_URL_ORIGIN`). No se ha ejercitado
|
||||
end-to-end un `POST /applications/*` contra el origen todavía — verificarlo en el
|
||||
próximo deploy antes de jubilar el flujo UI. El fix de raíz es corregir la regla
|
||||
del edge en el dashboard de Cloudflare.
|
||||
|
||||
### 11.2 La superficie real (del `openapi.yaml` del propio contenedor)
|
||||
|
||||
Dentro del contenedor: `/var/www/html/openapi.yaml`. Rutas declaradas en
|
||||
4.3.14 (extraídas el 2026-08-29):
|
||||
|
||||
- `/applications` + `/applications/{public, private-github-app, private-deploy-key, dockerfile, dockerimage}`
|
||||
- `/databases` + por motor: `{postgresql, clickhouse, dragonfly, redis, keydb, mariadb, mysql, mongodb}`
|
||||
- `/deployments`, `/deploy`, `/destinations`, `/servers` (+ `/import`, `/digitalocean`, `/hetzner`, `/vultr`)
|
||||
- `/github-apps`, `/gitlab-apps`, `/security/keys`, `/s3-storages`, `/tags`
|
||||
- `/notifications/{email, discord, slack, telegram, pushover, webhook}`
|
||||
- `/cloud-init-scripts`, `/cloud-tokens`
|
||||
- `/projects`, `/projects/{uuid}/environments`, `/resources`, `/services`
|
||||
- `/team`, `/teams`, `/team/envs`, `/team/members`
|
||||
- `/version`, `/health`, `/enable`, `/disable`, `/mcp/enable`, `/mcp/disable`
|
||||
|
||||
`GET /docs` (Scalar UI) redirige a `/login` — requiere sesión web; para el
|
||||
contrato exacto, leer el YAML del contenedor:
|
||||
`pct exec 102 -- docker exec coolify cat /var/www/html/openapi.yaml`.
|
||||
|
||||
### 11.3 POST-only confirmado como comportamiento de Coolify (no de Cloudflare)
|
||||
|
||||
Contra el origen, `GET /deploy?uuid=fake` → **405**
|
||||
`{"message":"This endpoint has changed to a POST request."}`. Es un cambio real
|
||||
de Coolify v4.2 (changelog: los endpoints de estado pasaron a POST-only), igual
|
||||
que §10.1. Otros cambios relevantes de 4.2→4.3: endpoints de logs para
|
||||
db/servicio/contenedor, settings de application incluidas en las respuestas,
|
||||
soporte MCP (read-only) y build pack Railpack (beta, 4.3.1). Breaking: se
|
||||
eliminó el endpoint deprecado de aplicación Docker Compose — usar
|
||||
`POST /services`. Fuentes: coolify.io/changelog y
|
||||
github.com/coollabsio/coolify/releases.
|
||||
|
||||
### 11.4 Estado de credenciales verificado el 2026-08-29
|
||||
|
||||
| Credencial | Estado |
|
||||
|---|---|
|
||||
| SSH `root@192.168.0.200` y `root@192.168.0.117` con `keys/proxmox_ed25519` (≡ `~/.ssh/coolify_key`, sin passphrase) | ✅ |
|
||||
| API Proxmox: token `root@pam!openclaw` (único en `token.cfg`) | ✅ 200 `/version` + `/cluster/resources` — ya en `.env.local.ps1` |
|
||||
| API Coolify: `COOLIFY_TOKEN` (team root) | ✅ 200 — v4.3.14 |
|
||||
| Gitea: token de `urieljareth` | ✅ 200 `/api/v1/user` |
|
||||
| GitHub: PAT `ghp_NFy4…` del `.env` | ❌ caducado (401) — reemplazado en `.env.local.ps1` por el token vivo del Credential Manager de Windows (`gho_…`, cuenta `urieljarethbusiness-cpu`, scopes `gist, repo, workflow`) |
|
||||
| Cloudflare API | ❌ sin token — gestionar túnel desde el dashboard |
|
||||
| UI Coolify: `urieljareth@gmail.com` / password habitual | ⚠️ probable (heredado de la instancia vieja `192.168.1.175`, ver `C:\Users\Uriel Jareth\coolify-agent\coolify-agent-skill.md`), sin verificar |
|
||||
|
||||
Inventario completo de llaves SSH de la máquina (incluidas las corruptas de
|
||||
SiteGround y la de la instancia antigua): `ACCESS.md` en la raíz del repo.
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
$env:PROXMOX_HOST = "192.168.0.200"
|
||||
$env:PROXMOX_NODE = "thinkcentre"
|
||||
$env:PROXMOX_USER = "root"
|
||||
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win"
|
||||
$env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519" # relativo a la raíz del repo (≡ ~\.ssh\coolify_key)
|
||||
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
|
||||
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
|
||||
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
|
||||
@@ -16,10 +16,10 @@ $env:PROXMOX_COOLIFY_LXC = "102"
|
||||
# Coolify
|
||||
$env:COOLIFY_API_URL = "https://coolify.urieljareth.org/api/v1"
|
||||
$env:COOLIFY_TOKEN = "REPLACE_WITH_COOLIFY_TOKEN"
|
||||
# Coolify UI login (email/password) — REQUIRED for Playwright-driven UI operations.
|
||||
# This instance (v4.1.2) does NOT expose the /applications/* REST API (all 404),
|
||||
# so configuring a git-based / Docker-Compose application (build pack, env vars,
|
||||
# domains) can ONLY be done through the web UI. See deploy_skill/references/coolify-4.1.2-notes.md.
|
||||
# Coolify UI login (email/password) — used by the Playwright UI fallback flow.
|
||||
# Since v4.3.14 the REST API is complete when called against the origin
|
||||
# (http://192.168.0.117:8000/api/v1); the /applications/* 404 happens only via
|
||||
# the public Cloudflare hostname. See deploy_skill/references/coolify-4.1.2-notes.md §11.
|
||||
$env:COOLIFY_EMAIL = "REPLACE_WITH_COOLIFY_UI_EMAIL"
|
||||
$env:COOLIFY_PASSWORD = "REPLACE_WITH_COOLIFY_UI_PASSWORD"
|
||||
|
||||
|
||||
@@ -29,7 +29,9 @@ $args = @("-sS", "-i", "-X", $Method.ToUpper(), "-H", "Authorization: Bearer $en
|
||||
if ($PSBoundParameters.ContainsKey("BodyJson")) {
|
||||
try { $null = $BodyJson | ConvertFrom-Json } catch { throw "BodyJson is not valid JSON: $($_.Exception.Message)" }
|
||||
$tmpBody = [System.IO.Path]::GetTempFileName()
|
||||
Set-Content -LiteralPath $tmpBody -Value $BodyJson -NoNewline -Encoding utf8
|
||||
# Set-Content -Encoding utf8 mete BOM en PS 5.1 y GitHub responde
|
||||
# 400 "Problems parsing JSON". WriteAllText con UTF8Encoding($false) no lo hace.
|
||||
[System.IO.File]::WriteAllText($tmpBody, $BodyJson, (New-Object System.Text.UTF8Encoding($false)))
|
||||
$args += @("--data", "@$tmpBody", "-H", "Content-Type: application/json")
|
||||
}
|
||||
|
||||
@@ -44,7 +46,9 @@ try {
|
||||
$bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" }
|
||||
|
||||
$statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim()
|
||||
if ($statusLine -notmatch " 20[04-9] ") {
|
||||
# 2xx completo: crear un repo responde 201 y la clase anterior ([04-9]) lo
|
||||
# trataba como error pese al éxito.
|
||||
if ($statusLine -notmatch " 20[0-9] ") {
|
||||
$snippet = $bodyBlock.Trim()
|
||||
if ($snippet.Length -gt 400) { $snippet = $snippet.Substring(0, 400) + "..." }
|
||||
throw "GitHub API HTTP error: $statusLine`nBody: $snippet"
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
<#
|
||||
.SYNOPSIS
|
||||
Create a Coolify application directly in the database (bypasses the 404 API and the web UI).
|
||||
|
||||
.DESCRIPTION
|
||||
For instances where POST /applications/* returns 404 and no UI credentials are available.
|
||||
Requires SSH access to the Proxmox host and the Coolify DB container.
|
||||
|
||||
Creates:
|
||||
1. applications row (Dockerfile build pack, GitHub source)
|
||||
2. application_settings row (prevents "disable_build_cache on null" crash)
|
||||
Then triggers a deploy via the API.
|
||||
|
||||
See deploy_skill/references/coolify-4.1.2-notes.md §8 for the full backstory.
|
||||
|
||||
.PARAMETER AppName
|
||||
Display name (also used in the generated UUID suffix).
|
||||
|
||||
.PARAMETER GitRepo
|
||||
GitHub repo in owner/repo format (e.g. urieljarethbusiness-cpu/agendamax).
|
||||
|
||||
.PARAMETER Fqdn
|
||||
Full domain (e.g. https://agendamax.urieljareth.org).
|
||||
|
||||
.PARAMETER Port
|
||||
Internal port the app listens on. Default: 3000.
|
||||
|
||||
.PARAMETER EnvironmentId
|
||||
Numeric ID of the Coolify environment. Find with:
|
||||
SELECT id FROM environments WHERE uuid = '<env_uuid>';
|
||||
|
||||
.PARAMETER GithubRepoId
|
||||
Numeric GitHub repo ID. Get from: GET https://api.github.com/repos/<owner>/<repo> → .id
|
||||
|
||||
.EXAMPLE
|
||||
. .\.env.local.ps1
|
||||
.\deploy_skill\scripts\New-CoolifyAppViaDB.ps1 `
|
||||
-AppName agendamax `
|
||||
-GitRepo urieljarethbusiness-cpu/agendamax `
|
||||
-Fqdn https://agendamax.urieljareth.org `
|
||||
-Port 3000 -EnvironmentId 9 -GithubRepoId 1314049704
|
||||
#>
|
||||
param(
|
||||
[Parameter(Mandatory)] [string]$AppName,
|
||||
[Parameter(Mandatory)] [string]$GitRepo,
|
||||
[Parameter(Mandatory)] [string]$Fqdn,
|
||||
[int]$Port = 3000,
|
||||
[Parameter(Mandatory)] [int]$EnvironmentId,
|
||||
[Parameter(Mandatory)] [int64]$GithubRepoId
|
||||
)
|
||||
|
||||
$ErrorActionPreference = "Stop"
|
||||
. "$PSScriptRoot\..\..\scripts\ProxmoxAgent.ps1"
|
||||
|
||||
# Generate a 24-char lowercase UUID (Coolify style)
|
||||
$uuid = -join ((1..24) | ForEach-Object { '{0:x}' -f (Get-Random -Max 16) })
|
||||
Write-Host "Generated UUID: $uuid" -ForegroundColor Cyan
|
||||
|
||||
# Build the SQL (single-line to avoid heredoc issues through SSH)
|
||||
$sql = @"
|
||||
INSERT INTO applications (uuid, name, git_repository, git_branch, git_commit_sha, build_pack, dockerfile_location, ports_exposes, health_check_path, health_check_port, health_check_host, health_check_method, health_check_return_code, health_check_scheme, health_check_interval, health_check_timeout, health_check_retries, health_check_start_period, health_check_enabled, limits_memory, limits_memory_swap, limits_memory_swappiness, limits_memory_reservation, limits_cpus, limits_cpu_shares, status, preview_url_template, fqdn, repository_project_id, source_type, source_id, destination_type, destination_id, environment_id, base_directory, static_image, created_at, updated_at) VALUES ('${uuid}', '${AppName}:main-${uuid}', '${GitRepo}', 'main', 'HEAD', 'dockerfile', '/Dockerfile', '${Port}', '/', '${Port}', 'localhost', 'GET', 200, 'http', 5, 5, 10, 5, false, '0', '0', 60, '0', '0', 1024, 'exited', '{{pr_id}}.{{domain}}', '${Fqdn}', ${GithubRepoId}, 'App\Models\GithubApp', 3, 'App\Models\StandaloneDocker', 0, ${EnvironmentId}, '/', 'nginx:alpine', NOW(), NOW()) RETURNING id;
|
||||
"@
|
||||
|
||||
# Write the SQL to a temp .sh script, scp it, execute it
|
||||
$tmpSh = [IO.Path]::GetTempFileName() + ".sh"
|
||||
@"
|
||||
pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -t -A -c "$($sql -replace '"', '\"' -replace "'", "'\''")"
|
||||
"@ | Set-Content -Path $tmpSh -Encoding ASCII
|
||||
|
||||
Write-Host "`n>>> Step 1: INSERT application row" -ForegroundColor Magenta
|
||||
$remoteSh = "/tmp/coolify_create_$(Get-Random).sh"
|
||||
scp -o BatchMode=yes -o StrictHostKeyChecking=no -i $env:PROXMOX_SSH_KEY $tmpSh "root@$($env:PROXMOX_HOST):$remoteSh" 2>&1 | Out-Null
|
||||
$appId = Invoke-ProxmoxSshCommand -Command "sh $remoteSh; rm $remoteSh"
|
||||
$appId = $appId.Trim()
|
||||
Remove-Item $tmpSh -ErrorAction SilentlyContinue
|
||||
Write-Host "Application ID: $appId" -ForegroundColor Green
|
||||
|
||||
if (-not $appId -or $appId -notmatch '^\d+$') {
|
||||
throw "INSERT failed or didn't return an ID. Output: $appId"
|
||||
}
|
||||
|
||||
# Step 2: application_settings
|
||||
Write-Host "`n>>> Step 2: INSERT application_settings row" -ForegroundColor Magenta
|
||||
$sqlSettings = @"
|
||||
INSERT INTO application_settings (application_id, is_static, is_git_submodules_enabled, is_git_lfs_enabled, is_auto_deploy_enabled, is_force_https_enabled, is_debug_enabled, is_preview_deployments_enabled, is_log_drain_enabled, is_gpu_enabled, is_swarm_only_worker_nodes, is_raw_compose_deployment_enabled, is_build_server_enabled, is_consistent_container_name_enabled, is_gzip_enabled, is_stripprefix_enabled, is_container_label_escape_enabled, is_container_label_readonly_enabled, disable_build_cache, is_spa, is_git_shallow_clone_enabled, is_pr_deployments_public_enabled, use_build_secrets, inject_build_args_to_dockerfile, docker_images_to_keep, created_at, updated_at) VALUES (${appId}, false, true, true, true, true, false, false, false, false, true, false, false, false, true, true, true, true, false, false, true, false, false, true, 2, NOW(), NOW());
|
||||
"@
|
||||
|
||||
$tmpSh2 = [IO.Path]::GetTempFileName() + ".sh"
|
||||
@"
|
||||
pct exec 102 -- docker exec coolify-db psql -U coolify -d coolify -c "$($sqlSettings -replace '"', '\"' -replace "'", "'\''")"
|
||||
"@ | Set-Content -Path $tmpSh2 -Encoding ASCII
|
||||
$remoteSh2 = "/tmp/coolify_settings_$(Get-Random).sh"
|
||||
scp -o BatchMode=yes -o StrictHostKeyChecking=no -i $env:PROXMOX_SSH_KEY $tmpSh2 "root@$($env:PROXMOX_HOST):$remoteSh2" 2>&1 | Out-Null
|
||||
Invoke-ProxmoxSshCommand -Command "sh $remoteSh2; rm $remoteSh2" | Write-Host
|
||||
Remove-Item $tmpSh2 -ErrorAction SilentlyContinue
|
||||
Write-Host "Settings created for app_id=$appId" -ForegroundColor Green
|
||||
|
||||
# Step 3: Deploy via API
|
||||
Write-Host "`n>>> Step 3: Trigger deploy" -ForegroundColor Magenta
|
||||
$headers = @{ Authorization = "Bearer $($env:COOLIFY_TOKEN)" }
|
||||
$deploy = Invoke-RestMethod "$($env:COOLIFY_API_URL)/deploy?uuid=${uuid}&force=true" -Headers $headers
|
||||
$deployUuid = $deploy.deployments[0].deployment_uuid
|
||||
Write-Host "Deployment queued: $deployUuid" -ForegroundColor Green
|
||||
|
||||
Write-Host "`n=== DONE ===" -ForegroundColor Green
|
||||
Write-Host "App UUID: $uuid"
|
||||
Write-Host "App ID: $appId"
|
||||
Write-Host "FQDN: $Fqdn"
|
||||
Write-Host "Deploy: $deployUuid"
|
||||
Write-Host "Monitor: GET $($env:COOLIFY_API_URL)/deployments/$deployUuid"
|
||||
@@ -213,13 +213,13 @@ Write-Host "Using server: $serverUuid" -ForegroundColor Cyan
|
||||
|
||||
if ($ServiceUuid) {
|
||||
Write-Host "Updating existing service: $ServiceUuid" -ForegroundColor Cyan
|
||||
# PATCH /services/{uuid} rechaza (422 "This field is not allowed")
|
||||
# server_uuid/project_uuid/environment_name — solo se envian name,
|
||||
# docker_compose_raw y urls. Verificado contra la instancia el 2026-09-02.
|
||||
$patch = [ordered]@{
|
||||
name = $AppName
|
||||
docker_compose_raw = $composeRaw
|
||||
docker_compose_raw = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($composeRaw))
|
||||
urls = $urlsList
|
||||
project_uuid = $proj.uuid
|
||||
environment_name = $env.name
|
||||
server_uuid = $serverUuid
|
||||
}
|
||||
if (-not (Confirm-Step "PATCH /services/$ServiceUuid (compose=$($composePath), urls=$($urlsList.Count), fqdn=$Fqdn)")) {
|
||||
throw "Aborted by user."
|
||||
@@ -227,13 +227,18 @@ if ($ServiceUuid) {
|
||||
$result = & $coolifyApi -Method PATCH -Path "/services/$ServiceUuid" -BodyJson ($patch | ConvertTo-Json -Depth 6) -Raw -ErrorAction Stop
|
||||
$svcUuid = $ServiceUuid
|
||||
} else {
|
||||
# Verified against Coolify 4.x on 2026-08-24:
|
||||
# - Do NOT send `type` together with `docker_compose_raw`. The API answers
|
||||
# 422 "You cannot provide both service type and docker_compose_raw."
|
||||
# `type` is only for one-click services from the library.
|
||||
# - `docker_compose_raw` MUST be base64. Sent raw it answers
|
||||
# 422 "The docker_compose_raw should be base64 encoded."
|
||||
$createBody = [ordered]@{
|
||||
type = "one-click-service" # custom service; Coolify accepts any non-empty string here
|
||||
name = $AppName
|
||||
project_uuid = $proj.uuid
|
||||
environment_name = $env.name
|
||||
server_uuid = $serverUuid
|
||||
docker_compose_raw = $composeRaw
|
||||
docker_compose_raw = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($composeRaw))
|
||||
urls = $urlsList
|
||||
instant_deploy = [bool]$InstantDeploy
|
||||
}
|
||||
|
||||
@@ -94,7 +94,7 @@ Each rule lists **what**, **why**, and **how to verify**.
|
||||
- **Why:** TLS is issued by Traefik via **DNS challenge** (Cloudflare API token), so it
|
||||
works regardless of Cloudflare's "Always Use HTTPS". Do **not** assume HTTP-01 challenge
|
||||
— that path is intentionally not used here
|
||||
(see [`issue-coolify-static-app-deploy.md`](issue-coolify-static-app-deploy.md)).
|
||||
(see [`2026-04-11-coolify-static-app-deploy.md`](incidentes/2026-04-11-coolify-static-app-deploy.md)).
|
||||
- **Verify:**
|
||||
```powershell
|
||||
curl.exe -k -sSI https://<name>.urieljareth.org/
|
||||
@@ -285,13 +285,13 @@ curl.exe -k -sSI https://<name>.urieljareth.org/
|
||||
|
||||
| Symptom | Root cause | Fix | Source |
|
||||
|---------|-----------|-----|--------|
|
||||
| App's domain returns **502 Bad Gateway** | `ports_exposes` doesn't match the real listening port (often `3000` vs `80`) | Set `ports_exposes` to the actual port before deploy | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) |
|
||||
| 502 / no certificate on the app domain | Traefik couldn't get a cert via HTTP challenge | Environment uses **DNS challenge**; ensure FQDN is set and don't depend on HTTP-01 | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) |
|
||||
| App reachable only at an ugly **UUID subdomain** | FQDN not set, Coolify auto-generated it | Set `fqdn` to `https://<name>.urieljareth.org` before first deploy | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) |
|
||||
| App's domain returns **502 Bad Gateway** | `ports_exposes` doesn't match the real listening port (often `3000` vs `80`) | Set `ports_exposes` to the actual port before deploy | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
|
||||
| 502 / no certificate on the app domain | Traefik couldn't get a cert via HTTP challenge | Environment uses **DNS challenge**; ensure FQDN is set and don't depend on HTTP-01 | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
|
||||
| App reachable only at an ugly **UUID subdomain** | FQDN not set, Coolify auto-generated it | Set `fqdn` to `https://<name>.urieljareth.org` before first deploy | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
|
||||
| App **won't start**, port conflict | Compose publishes `80`/`443` (owned by `coolify-proxy`) | Remove host port publishing; let Traefik route | [runbooks/baserow.md](runbooks/baserow.md) |
|
||||
| App **can't reach its DB** | Used `localhost`, or DB is on a different network | Use the DB **service name**; for shared services join the `coolify` network | [runbooks/nextcloud.md](runbooks/nextcloud.md), [runbooks/baserow.md](runbooks/baserow.md) |
|
||||
| Coolify **UI blank** when opening the app page | Cloudflare tunnel route order — `/app/*` captured `/application/...` | Operator fix: `/project/*` route must precede `/app/*` in the dashboard | [issue-coolify-static-app-deploy.md](issue-coolify-static-app-deploy.md) |
|
||||
| **WebSocket / terminal** drops, `tls: first record does not look like a TLS handshake` | Tunnel routes for ports `6001`/`6002` set to `https://` | Operator fix: those routes must be `http://` | [ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md](ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md) |
|
||||
| Coolify **UI blank** when opening the app page | Cloudflare tunnel route order — `/app/*` captured `/application/...` | Operator fix: `/project/*` route must precede `/app/*` in the dashboard | [2026-04-11-coolify-static-app-deploy.md](incidentes/2026-04-11-coolify-static-app-deploy.md) |
|
||||
| **WebSocket / terminal** drops, `tls: first record does not look like a TLS handshake` | Tunnel routes for ports `6001`/`6002` set to `https://` | Operator fix: those routes must be `http://` | [2026-04-11-cloudflare-tunnel-websocket-tls.md](incidentes/2026-04-11-cloudflare-tunnel-websocket-tls.md) |
|
||||
|
||||
> The last two are *operator/infrastructure* fixes (Cloudflare dashboard), not things the
|
||||
> app developer changes — listed here so an agent recognizes the symptom and points the
|
||||
|
||||
@@ -0,0 +1,408 @@
|
||||
# Tool index — catálogo canónico de herramientas
|
||||
|
||||
**Este es el índice único de todo lo ejecutable del repo.** Si buscas "qué script
|
||||
uso para X", empieza aquí y no en los `TOOLS.md` de cada skill (esos son guías de
|
||||
uso; este es el catálogo).
|
||||
|
||||
Verificado contra el host real el **2026-08-07**. Las firmas de parámetros se
|
||||
extrajeron del AST de PowerShell, no a mano.
|
||||
|
||||
---
|
||||
|
||||
## 0. Cómo leer este índice
|
||||
|
||||
- **R/W** — `RO` = solo lectura, se puede ejecutar sin preguntar. `W` = muta
|
||||
estado, **exige confirmación explícita del usuario antes de ejecutar**.
|
||||
`RO/W` = depende de los parámetros (columna "notas" lo aclara).
|
||||
- **Env** — variables que deben estar cargadas (`. .\.env.local.ps1`). Si faltan,
|
||||
el script *lanza excepción*, no falla silenciosamente.
|
||||
- Todo se ejecuta desde la raíz del repo, en PowerShell.
|
||||
|
||||
---
|
||||
|
||||
## 1. Los 6 gotchas que producen resultados silenciosamente incorrectos
|
||||
|
||||
Léelos antes de invocar nada. No son teóricos: los cuatro primeros se
|
||||
verificaron el 2026-08-07 y el quinto el 2026-08-23. Cada uno rompe una tarea de
|
||||
forma que *parece* haber funcionado.
|
||||
|
||||
### 1.1 `-Raw` significa lo OPUESTO en dos familias de wrappers
|
||||
|
||||
Hay dos convenciones incompatibles. Elegir mal no da error: da un resultado vacío.
|
||||
|
||||
| Wrapper | Sin `-Raw` (default) | Con `-Raw` |
|
||||
|---|---|---|
|
||||
| `coolify_skill\scripts\Invoke-CoolifyApi.ps1` | **string JSON** | **objetos PowerShell** |
|
||||
| `scripts\Invoke-CloudflareApi.ps1` | **string JSON** | **objetos PowerShell** |
|
||||
| `deploy_skill\scripts\Invoke-GitHubApi.ps1` | **objetos PowerShell** | `{Status, Headers, Body}` |
|
||||
| `gitea_skill\scripts\Invoke-GiteaApi.ps1` | **objetos PowerShell** | `{Status, Headers, Body}` |
|
||||
|
||||
Consecuencia práctica: en Coolify y Cloudflare, **si vas a filtrar o proyectar el
|
||||
resultado necesitas `-Raw`**. Sin él recibes un `System.String` y
|
||||
|
||||
```powershell
|
||||
# MAL: devuelve una fila vacía, sin error. El resultado es un string.
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" |
|
||||
Select-Object name, uuid
|
||||
|
||||
# BIEN
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
|
||||
Select-Object name, uuid
|
||||
```
|
||||
|
||||
Usa el default (string JSON) solo cuando vas a mostrar la respuesta tal cual.
|
||||
|
||||
### 1.2 `Invoke-ProxmoxSsh.ps1` corrompe las comillas anidadas
|
||||
|
||||
`Invoke-ProxmoxSshCommand` pasa `$Command` como un único argumento a `ssh`, y
|
||||
PowerShell 5.1 destroza las comillas embebidas al invocar un ejecutable nativo.
|
||||
Cualquier comando con quoting anidado —típicamente `docker exec ... bash -lc '...
|
||||
psql -c "SELECT ..."'`— llega mutilado al host:
|
||||
|
||||
```
|
||||
bash: line 1: -c: command not found
|
||||
psql: option requires an argument -- 'F'
|
||||
```
|
||||
|
||||
**Solución verificada: codifica el comando remoto en base64.** Es el patrón a
|
||||
usar para cualquier cosa con más de un nivel de comillas:
|
||||
|
||||
```powershell
|
||||
$remote = @'
|
||||
pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -c "SELECT name FROM public.installation_configs"'
|
||||
'@
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($remote))
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "echo $b64 | base64 -d | bash 2>&1"
|
||||
```
|
||||
|
||||
El here-string `@'...'@` (comillas simples) es obligatorio: evita que PowerShell
|
||||
expanda `$POSTGRES_PASSWORD` del lado de Windows.
|
||||
|
||||
Para comandos de un solo nivel de comillas (`pct list`, `docker ps --format
|
||||
'{{.Names}}'`) el wrapper directo funciona bien.
|
||||
|
||||
### 1.3 Los nombres de contenedor NO se pueden adivinar
|
||||
|
||||
Coolify nombra cada contenedor `<servicio>-<uuid>` (y a veces le añade un sufijo
|
||||
numérico de build). No existe un contenedor llamado `chatwoot` ni `nextcloud`:
|
||||
|
||||
```
|
||||
chatwoot-c11xzy2tx2cdapm32f5b89vy
|
||||
nextcloud-db-hdcdpkm0jko3qqvn5683ercc
|
||||
web-instademo0portal0insta0demo1-060825532589
|
||||
```
|
||||
|
||||
**Siempre resuelve el nombre real antes de usarlo.** Dos caminos:
|
||||
|
||||
```powershell
|
||||
# a) desde Docker, por patrón
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}' | grep -i chatwoot"
|
||||
|
||||
# b) desde Coolify, para obtener el uuid del recurso (y de ahí el sufijo)
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
|
||||
Where-Object { $_.name -match 'chatwoot' } | Select-Object name, uuid, fqdn
|
||||
```
|
||||
|
||||
Corolario: **un redeploy puede recrear el contenedor con otro sufijo**, y cualquier
|
||||
script o cron que tenga el nombre hardcodeado empieza a fallar. Es exactamente lo
|
||||
que le pasó al guard de Chatwoot (ver [runbooks/chatwoot-update.md](runbooks/chatwoot-update.md)).
|
||||
|
||||
### 1.4 `/applications/*` da 404 — por Cloudflare, no por Coolify (corregido 2026-08-29)
|
||||
|
||||
**Re-verificado a fondo el 2026-08-29 sobre v4.3.14 y el diagnóstico anterior
|
||||
cambió:** el 404 NO lo produce Coolify. Es un bloqueo del edge de Cloudflare en
|
||||
el hostname público. Mismo token, misma ruta:
|
||||
|
||||
| Llamada | Resultado |
|
||||
|---|---|
|
||||
| `https://coolify.urieljareth.org/api/v1/applications` (vía Cloudflare) | **404** |
|
||||
| `http://192.168.0.117:8000/api/v1/applications` (origen, LXC 102) | **200** |
|
||||
| `/github-apps` | igual: 404 vía CF, 200 vía origen |
|
||||
| `/version`, `/resources`, `/services`, `/databases`, `/projects`, `/servers`, `/teams`, `/deployments`, `/security/keys` | OK por ambas vías |
|
||||
|
||||
La API del origen está **completa**: el `openapi.yaml` del contenedor
|
||||
(`/var/www/html/openapi.yaml`) declara todo el namespace de `/applications/*`,
|
||||
notificaciones, proveedores cloud y MCP, y responde. **Para esos endpoints,
|
||||
apunta la llamada al origen** (`$env:COOLIFY_API_URL_ORIGIN`) o corrige la regla
|
||||
de Cloudflare en el dashboard. Además, desde v4.2 los endpoints de estado
|
||||
exigen **POST** (`GET /deploy` → 405; ver notas §10.1).
|
||||
|
||||
Esto re-habilita (previa verificación en el próximo deploy) herramientas que
|
||||
estaban marcadas rotas — ver §4.
|
||||
|
||||
**Nota (2026-08-24):** `POST /services` **sí funciona** para stacks compose
|
||||
propios, con tres condiciones que la API no perdona: **no** enviar `type` junto a
|
||||
`docker_compose_raw`, mandar el compose en **base64**, y enviar el body como
|
||||
**bytes UTF-8**. Las tres estaban mal en el toolkit y ya están corregidas; el
|
||||
detalle está en
|
||||
[docs/casos/firecrawl-stack-minimo.md §4](casos/firecrawl-stack-minimo.md).
|
||||
`Test-PreDeployChecklist.ps1` sigue sin leer `docker-compose.coolify.yml`
|
||||
(solo mira `docker-compose.yml|yaml` y `compose.yml|yaml`), así que **puede dar
|
||||
todo PASS sobre un archivo que no abrió**.
|
||||
|
||||
### 1.5 "Verde en Coolify" NO significa "enrutado en Traefik"
|
||||
|
||||
Verificado el 2026-08-23. Es la causa de que un servicio recién creado desde la
|
||||
librería de Coolify devuelva **`503 no available server`** estando en verde.
|
||||
|
||||
Son dos señales distintas y la UI solo muestra una:
|
||||
|
||||
| Señal | Quién la usa | Qué significa |
|
||||
|---|---|---|
|
||||
| `State.Status = running` | **La UI de Coolify** (el punto verde) | El contenedor existe y no ha muerto |
|
||||
| `State.Health.Status = healthy` | **Traefik** | El contenedor entra al balanceador |
|
||||
|
||||
Traefik solo enruta contenedores que Docker reporta `healthy`. Un contenedor
|
||||
`running` + `unhealthy` **no tiene ruta**, la petición cae al catch-all de
|
||||
Coolify (`default_redirect_503.yaml`: `priority: -1000`, servicio `noop` con
|
||||
`servers: { }`) y de ahí sale el string `no available server`.
|
||||
|
||||
En este host el primer boot de un servicio tarda **minutos** (rootfs ext4 sobre
|
||||
loopback sobre HDD, ~39 ms por escritura), pero **12 de los 14 servicios no
|
||||
tienen `start_period`** en su healthcheck. Se les declara `unhealthy` mucho antes
|
||||
de que la app llegue a escuchar.
|
||||
|
||||
**Distingue el código HTTP antes de tocar nada:**
|
||||
|
||||
- **`502 Bad Gateway`** → Traefik *tiene* la ruta, el backend rechaza. Problema
|
||||
de la app o del puerto.
|
||||
- **`503 no available server`** → Traefik **no tiene** la ruta. Casi siempre es
|
||||
un contenedor que todavía está arrancando. **Espera, no redeployes**: un
|
||||
redeploy reinicia el entrypoint desde cero y reinicia el arranque lento.
|
||||
|
||||
```powershell
|
||||
# El diagnóstico correcto, en un comando (solo lectura):
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
|
||||
```
|
||||
|
||||
Detalle completo, evidencia e hipótesis descartadas:
|
||||
[docs/casos/coolify-servicio-nuevo-503-no-available-server.md](casos/coolify-servicio-nuevo-503-no-available-server.md).
|
||||
|
||||
### 1.6 Si la imagen no expone el puerto donde sirve, el FQDN necesita `:puerto`
|
||||
|
||||
Verificado el 2026-08-24 diagnosticando `grimmory`, que devolvía **502**.
|
||||
|
||||
Coolify deriva el label `traefik.http.services.*.loadbalancer.server.port` del
|
||||
**puerto que lleve el FQDN guardado** en `service_applications.fqdn`. Si el FQDN
|
||||
no lleva puerto, Coolify **no emite el label**, y Traefik cae al único puerto que
|
||||
declare la imagen (`Config.ExposedPorts`).
|
||||
|
||||
Eso funciona por accidente cuando app e imagen coinciden, y falla en silencio
|
||||
cuando no:
|
||||
|
||||
| Servicio | FQDN guardado | Sirve en | Imagen expone | Resultado |
|
||||
|---|---|---|---|---|
|
||||
| `nextcloud` | sin puerto | 80 | 80 | OK por coincidencia |
|
||||
| `n8n` | `…:5678` | 5678 | 5678 | OK explícito |
|
||||
| `qdrant` | `…:6333` | 6333 | 6333 | OK explícito |
|
||||
| **`grimmory`** | **sin puerto** | **80** | **6060** | **502** |
|
||||
|
||||
`grimmory` corría `healthy` (su healthcheck prueba `http://127.0.0.1/health`, o
|
||||
sea el 80, y pasaba), Traefik lo enrutaba, y llegaba al 6060 donde no hay nada:
|
||||
**connection refused → 502**.
|
||||
|
||||
Ojo con la confusión: tener `SERVICE_URL_GRIMMORY_80` en el compose **no basta**.
|
||||
Lo que manda es el puerto en el FQDN almacenado.
|
||||
|
||||
**Cómo distinguirlo de otros fallos:**
|
||||
|
||||
- **`502`** → hay ruta, el backend rechaza. Compara el puerto donde escucha la app
|
||||
con el que busca Traefik. Casi siempre es esto.
|
||||
- **`503 no available server`** → no hay ruta (§1.5).
|
||||
|
||||
```powershell
|
||||
# Puertos donde escucha de verdad (en hex; 0050 = 80, 1F90 = 8080):
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'cat /proc/net/tcp'"
|
||||
|
||||
# Puerto que busca Traefik (si no sale nada, cae al ExposedPorts de la imagen):
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker inspect <cont> --format '{{json .Config.Labels}}'"
|
||||
```
|
||||
|
||||
**Arreglo:** poner el puerto en el FQDN (`https://x.urieljareth.org:80`) y
|
||||
**redeployar** el servicio — los labels solo se regeneran al recrear el
|
||||
contenedor.
|
||||
|
||||
---
|
||||
|
||||
## 2. Infraestructura: host Proxmox, red, apps del host
|
||||
|
||||
Todo llega al host por un solo camino:
|
||||
|
||||
```
|
||||
PowerShell → Invoke-ProxmoxSshCommand → ssh [email protected]
|
||||
→ pct exec 102 -- docker ... (para cualquier cosa de Docker)
|
||||
```
|
||||
|
||||
| Script | R/W | Env | Parámetros | Qué hace |
|
||||
|---|---|---|---|---|
|
||||
| [scripts/ProxmoxAgent.ps1](../scripts/ProxmoxAgent.ps1) | librería | — | — | **Dot-source obligatorio** (`. .\scripts\ProxmoxAgent.ps1`). Expone `Get-ProxmoxConfig`, `Assert-ProxmoxConfig`, `Invoke-ProxmoxSshCommand`, `Invoke-ProxmoxApi`. Todos los demás scripts lo consumen; no reimplementes la conexión. |
|
||||
| [scripts/Test-ProxmoxConnection.ps1](../scripts/Test-ProxmoxConnection.ps1) | RO | opcional `PROXMOX_API_TOKEN_*` | — | Smoke test: config + SSH + muestra de Docker + auth de API. Sale 1 si algo falla. **Ejecútalo antes de trabajo operativo.** |
|
||||
| [scripts/Get-ProxmoxInventory.ps1](../scripts/Get-ProxmoxInventory.ps1) | RO | — | — | Snapshot completo: host, LXC, QEMU, Docker en LXC 102. |
|
||||
| [scripts/Invoke-ProxmoxSsh.ps1](../scripts/Invoke-ProxmoxSsh.ps1) | **RO/W** | — | `-Command <string>` | Comando arbitrario por SSH. **El R/W lo determina el comando**: `pct list` es RO, `pct stop` es W. Ver gotcha §1.2 para comillas anidadas. |
|
||||
| [scripts/Invoke-CloudflareApi.ps1](../scripts/Invoke-CloudflareApi.ps1) | **RO/W** | `CLOUDFLARE_API_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Cloudflare (túnel + DNS). `GET` es RO; el resto W. `-Raw` → objetos (§1.1). |
|
||||
| [scripts/Install-CoolifyAutostart.ps1](../scripts/Install-CoolifyAutostart.ps1) | **W** (RO con `-VerifyOnly`) | — | `-VerifyOnly` `-SkipOnboot` `-RunNow` `-Uninstall` | Instala el auto-arranque del stack tras corte de luz (LXC 102 + túnel). Usa `-VerifyOnly` para auditar sin tocar nada. |
|
||||
|
||||
**API REST de Proxmox** (distinta de la de Coolify): requiere
|
||||
`PROXMOX_API_TOKEN_ID` + `PROXMOX_API_TOKEN_SECRET`. `Invoke-ProxmoxApi` lanza
|
||||
excepción si faltan.
|
||||
|
||||
> ⚠️ **Estado actual (2026-08-07):** `.env.local.ps1` **no** define
|
||||
> `PROXMOX_API_TOKEN_ID`/`_SECRET`, `CLOUDFLARE_API_TOKEN` ni
|
||||
> `COOLIFY_EMAIL`/`COOLIFY_PASSWORD`. Las tres rutas que dependen de ellas
|
||||
> (API de Proxmox, API de Cloudflare, flujo UI de Coolify) **fallan hoy**. Lo
|
||||
> que sí está cargado: `PROXMOX_HOST/NODE/USER/SSH_KEY/COOLIFY_LXC`,
|
||||
> `COOLIFY_API_URL`, `COOLIFY_TOKEN`, `GITHUB_TOKEN`, `GITHUB_OWNER`,
|
||||
> `GITEA_URL/USER/TOKEN`.
|
||||
|
||||
### 2.1 Chatwoot — parche enterprise
|
||||
|
||||
Contexto completo en [runbooks/chatwoot-update.md](runbooks/chatwoot-update.md).
|
||||
|
||||
| Script | R/W | Parámetros | Qué hace |
|
||||
|---|---|---|---|
|
||||
| [scripts/Get-ChatwootLicenseStatus.ps1](../scripts/Get-ChatwootLicenseStatus.ps1) | RO | `-ServiceUuid` `-AppContainer` `-DbContainer` `-Deep` | Estado de la licencia. `-Deep` verifica además los feature flags por cuenta (`accounts.feature_flags`). **Usa esto para diagnosticar; nunca el parche.** |
|
||||
| [scripts/Apply-ChatwootEnterprisePatch.ps1](../scripts/Apply-ChatwootEnterprisePatch.ps1) | **W** | `-DryRun` `-ReenableAccountFeatures` `-ServiceUuid` `-Container` `-AppContainer` `-LxcId` `-ProxmoxHost` `-SshKey` | Reaplica el parche. **El SQL de 3 filas no basta**: pasa siempre `-ReenableAccountFeatures`. Usa `-DryRun` primero. |
|
||||
| [scripts/chatwoot-enterprise-guard.sh](../scripts/chatwoot-enterprise-guard.sh) | — | — | **Copia versionada** del guard que corre en el host. No se ejecuta desde Windows. Instalado en `/root/scripts/chatwoot-enterprise-guard.sh`, agendado por `/etc/cron.d/chatwoot-enterprise-guard` cada 5 min. Log: `/var/log/chatwoot-enterprise-guard.log` (solo escribe cuando actúa). |
|
||||
|
||||
### 2.2 Artefactos que viven en el host (no se invocan desde Windows)
|
||||
|
||||
| Archivo | Qué es |
|
||||
|---|---|
|
||||
| [scripts/host/coolify-autostart.sh](../scripts/host/coolify-autostart.sh) | Script de arranque; lo despliega `Install-CoolifyAutostart.ps1`. |
|
||||
| [scripts/host/coolify-autostart.service](../scripts/host/coolify-autostart.service) | Unit de systemd correspondiente. |
|
||||
| [scripts/fix-nextcloud-config.php](../scripts/fix-nextcloud-config.php) | Fragmento puntual del fix HTTPS de Nextcloud. Ver [runbooks/nextcloud.md](runbooks/nextcloud.md). |
|
||||
| [scripts/Set-ProxmoxEnv.example.ps1](../scripts/Set-ProxmoxEnv.example.ps1) | Plantilla de entorno. El template completo es [.env.example](../.env.example). |
|
||||
|
||||
### 2.3 Scripts de deploy específicos de una app
|
||||
|
||||
`scripts/apps/` guarda scripts one-off con uuid y dominio **hardcodeados**. Son
|
||||
**mutantes** y están atados a un recurso concreto: lee la cabecera antes de
|
||||
ejecutar uno, y verifica que el uuid siga siendo el correcto (§1.3).
|
||||
|
||||
| Script | R/W | Env | Qué hace |
|
||||
|---|---|---|---|
|
||||
| [scripts/apps/Deploy-SoloLeveling.ps1](../scripts/apps/Deploy-SoloLeveling.ps1) | **W** | `COOLIFY_*`, `PROXMOX_*`, `GITHUB_TOKEN` (scope `repo`) | Re-deploy de "El Sistema (Solo Leveling)" sin depender del pull de GHCR (el registry es privado y el token no tiene `read:packages`). Construye la imagen **en el servidor** (clone del repo privado + `docker build`), la taguea con el nombre que espera el compose generado por Coolify, y levanta el servicio. Params: `-ServiceUuid` `-Domain` `-Repo` `-Image` `-NoBuild`. Requiere que el service ya esté registrado en Coolify. |
|
||||
| [scripts/apps/Deploy-OhDaddy.ps1](../scripts/apps/Deploy-OhDaddy.ps1) | **W** | `COOLIFY_*`, `PROXMOX_*` | Redeploy de oh-daddy (stack app+db+Inngest self-hosted, servicio `rzittzudkunwx8gilonn7tqe`). Clone del repo público + `stacks/oh-daddy/Dockerfile` inyectado, build de `oh-daddy-app:local` en el server, `compose up -d`, schema idempotente y re-registro de funciones Inngest (`PUT /api/inngest`). Params: `-ServiceUuid` `-Fqdn` `-Repo` `-Image` `-NoBuild`. Detalle: [casos/oh-daddy-deploy.md](casos/oh-daddy-deploy.md). |
|
||||
|
||||
---
|
||||
|
||||
## 3. Coolify — operación
|
||||
|
||||
| Script | R/W | Env | Parámetros | Qué hace |
|
||||
|---|---|---|---|---|
|
||||
| [coolify_skill/scripts/Get-CoolifyDockerStatus.ps1](../coolify_skill/scripts/Get-CoolifyDockerStatus.ps1) | RO | — | `-Filter <regex>` `-All` | Estado de contenedores vía LXC 102. `-All` lista todo; `-Filter` acota por regex. Primera parada para "¿está corriendo X?". |
|
||||
| [coolify_skill/scripts/Test-CoolifyServiceReady.ps1](../coolify_skill/scripts/Test-CoolifyServiceReady.ps1) | RO | — | `-Uuid <uuid>` `-Fqdn <url>` `-WaitSeconds <n>` | Contrasta `running` vs `healthy` por contenedor (la discrepancia que la UI esconde), avisa de `start_period` insuficiente, detecta setup en curso (`chown`/`apt`) y prueba el dominio distinguiendo 503 de 502. **Primera parada para un `503 no available server`** — ver §1.5. |
|
||||
| [coolify_skill/scripts/Set-CoolifyHealthcheckGrace.ps1](../coolify_skill/scripts/Set-CoolifyHealthcheckGrace.ps1) | **RO/W** | — | `-Uuid <uuid>` `-StartPeriodSeconds 300` `-MinIntervalSeconds 10` `-ShowResult` `-Apply` | Inserta `start_period` en los healthcheck que no lo tienen y sube intervalos demasiado cortos, editando `services.docker_compose_raw`. **Dry-run sin `-Apply`.** Con `-Apply` guarda copia de rollback en `backups/` y verifica leyendo de vuelta. **No redeploya** — el healthcheck solo aplica al recrear el contenedor. Ver §1.5. |
|
||||
| [coolify_skill/scripts/Invoke-CoolifyApi.ps1](../coolify_skill/scripts/Invoke-CoolifyApi.ps1) | **RO/W** | `COOLIFY_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Coolify. `GET` RO; `POST/PUT/PATCH/DELETE` W → confirmación. Ver §1.1 (`-Raw`) y §1.4 (endpoints que dan 404). |
|
||||
| `coolify_skill/scripts/coolify.sh` | RO/W | `COOLIFY_TOKEN` | — | Helper Bash legacy para sesiones Linux/WSL. En este repo se prefieren los wrappers PowerShell. |
|
||||
|
||||
**Referencia de la API:** no cargues el árbol completo. Busca y abre un solo
|
||||
archivo:
|
||||
|
||||
```powershell
|
||||
rg -n "deploy|database|environment" .\coolify_skill\references
|
||||
Get-Content .\coolify_skill\references\ops\deploy-by-tag-or-uuid.md
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Deploy de proyectos nuevos a Coolify
|
||||
|
||||
> ⚠️ **Lee esto antes de usar cualquier cosa de esta sección.** El pipeline
|
||||
> "una sola línea" fallaba porque `/applications/*` daba 404 (§1.4). Desde el
|
||||
> 2026-08-29 sabemos que ese 404 es de **Cloudflare, no de Coolify**: apuntando
|
||||
> `COOLIFY_API_URL` al origen (`http://192.168.0.117:8000/api/v1`) el namespace
|
||||
> completo responde. Scripts marcados ❌ abajo deben funcionar así — pendiente de
|
||||
> verificar en el próximo deploy; el flujo UI y la vía DB siguen como fallback.
|
||||
|
||||
| Script | R/W | Env | Estado en esta instancia |
|
||||
|---|---|---|---|
|
||||
| [Publish-ProjectToCoolify.ps1](../deploy_skill/scripts/Publish-ProjectToCoolify.ps1) | W | `GITHUB_TOKEN`, `COOLIFY_TOKEN` | ⚠️ Su paso 5 invoca `New-CoolifyApplication.ps1` → probar contra el origen. |
|
||||
| [New-CoolifyApplication.ps1](../deploy_skill/scripts/New-CoolifyApplication.ps1) | W | `COOLIFY_TOKEN` | ⚠️ Usa `POST /applications/public` y `PATCH /applications/{uuid}` → funcionan vía origen (404 solo vía Cloudflare). |
|
||||
| [Invoke-CoolifyRollback.ps1](../deploy_skill/scripts/Invoke-CoolifyRollback.ps1) | W | `COOLIFY_TOKEN` | ⚠️ Usa `PATCH /applications/{uuid}` → funciona vía origen. |
|
||||
| [New-CoolifyService.ps1](../deploy_skill/scripts/New-CoolifyService.ps1) | W | `COOLIFY_TOKEN` | ✅ **Funciona.** Usa `POST /services`. Es la vía válida para stacks multi-contenedor. |
|
||||
| [New-CoolifyAppViaDB.ps1](../deploy_skill/scripts/New-CoolifyAppViaDB.ps1) | **W (INSERT directo en la DB)** | — | ⚠️ **Último recurso.** Escribe la fila en `applications` de la DB de Coolify saltándose API y UI. Sin validación ni rollback. Requiere `-EnvironmentId` y `-GithubRepoId` reales. |
|
||||
| `coolify-ui/coolify-login.mjs` + `Configure-CoolifyComposeApp.mjs` | W (vía navegador) | `COOLIFY_EMAIL`, `COOLIFY_PASSWORD` | ✅ Fallback soportado para apps git build-from-source. Credenciales probables ya en `.env.local.ps1` (sin verificar). |
|
||||
|
||||
**Vías, en orden de preferencia:**
|
||||
|
||||
1. Stack multi-contenedor con `docker-compose.coolify.yml` → `New-CoolifyService.ps1` (API).
|
||||
2. App git build-from-source → API contra el origen; si falla, flujo UI con Playwright (`coolify-ui/`).
|
||||
3. Último recurso → `New-CoolifyAppViaDB.ps1`.
|
||||
|
||||
Detalle completo del flujo v4.1.2 y sus trampas (BOM UTF-8 rompe el parser YAML
|
||||
de Coolify; "Reload Compose File" es obligatorio; fijar dominios por servicio o
|
||||
sale 503; no encolar deploys concurrentes) en
|
||||
[deploy_skill/references/coolify-4.1.2-notes.md](../deploy_skill/references/coolify-4.1.2-notes.md).
|
||||
|
||||
### 4.1 Scripts de deploy que funcionan sin depender de `/applications`
|
||||
|
||||
| Script | R/W | Env | Parámetros | Qué hace |
|
||||
|---|---|---|---|---|
|
||||
| [Initialize-CoolifyProject.ps1](../deploy_skill/scripts/Initialize-CoolifyProject.ps1) | W (solo local) | — | `-Path` `-Stack {node\|python\|compose\|static}` `-AppPort` `-Force` | Genera Dockerfile/compose/.dockerignore compatibles. Idempotente: no sobreescribe sin `-Force`. |
|
||||
| [Test-PreDeployChecklist.ps1](../deploy_skill/scripts/Test-PreDeployChecklist.ps1) | RO | — | `-Path` `-ExpectedPort` `-Strict` | Valida el proyecto contra las reglas duras de [AGENTS-coolify-apps.md](AGENTS-coolify-apps.md). **Gate obligatorio antes de cualquier push.** |
|
||||
| [New-GitHubRepo.ps1](../deploy_skill/scripts/New-GitHubRepo.ps1) | W (remoto) | `GITHUB_TOKEN` | `-Name` `-Description` `-Private` `-Owner` | Crea el repo en GitHub. Idempotente: avisa si ya existe. |
|
||||
| [Invoke-GitHubApi.ps1](../deploy_skill/scripts/Invoke-GitHubApi.ps1) | RO/W | `GITHUB_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de GitHub. Default → objetos (§1.1). |
|
||||
| [Test-PostDeploy.ps1](../deploy_skill/scripts/Test-PostDeploy.ps1) | RO | `COOLIFY_TOKEN` (opcional) | `-Fqdn` `-ContainerName` `-SiblingService` `-ApplicationUuid` | Verifica el estado en Coolify: contenedores, logs del proxy, DNS entre hermanos. `-ApplicationUuid` consulta `/applications/*` → vía pública da 404 (Cloudflare), vía origen funciona. |
|
||||
| [Test-ServiceOnline.ps1](../deploy_skill/scripts/Test-ServiceOnline.ps1) | RO | — | `-Fqdn` `-Path` `-ExpectedStatusCode` `-ExpectTitle` `-Screenshot` `-SkipBrowser` `-TimeoutMs` `-Retries` | **La "definición de done".** Dos capas: HTTP 200 vía curl + render real en Chromium (Playwright). Detecta 502 de Traefik, TLS a medias y crashes de JS del cliente. Sin Node/Playwright la capa de navegador se omite con warning, no falla. |
|
||||
| `deploy_skill/scripts/verify-online.mjs` | RO | — | (invocado por el script de arriba) | Capa de navegador de `Test-ServiceOnline.ps1`: navegación real con Playwright, captura `pageerror` y requests fallidos del main frame. No lo invoques directo. Requiere `npm install` en la raíz del repo. |
|
||||
|
||||
`git push` usa **Windows Credential Manager (wincred)**, no el PAT. Verificado
|
||||
para la cuenta `urieljarethbusiness-cpu`.
|
||||
|
||||
**Plantillas:** [deploy_skill/references/templates/](../deploy_skill/references/templates/) —
|
||||
`Dockerfile.node`, `Dockerfile.python`, `docker-compose.app-db.yml`,
|
||||
`env.local.template.ps1`.
|
||||
|
||||
---
|
||||
|
||||
## 5. Gitea — hosting git self-hosted
|
||||
|
||||
Distinto de deploy_skill: esto administra la capa de hosting git, no Coolify.
|
||||
El propio repo Manager vive aquí.
|
||||
|
||||
| Script | R/W | Env | Parámetros | Qué hace |
|
||||
|---|---|---|---|---|
|
||||
| [Test-GiteaConnection.ps1](../gitea_skill/scripts/Test-GiteaConnection.ps1) | RO | `GITEA_URL`, `GITEA_TOKEN` | — | Smoke test: `/version`, `/user`, `/settings/api`, `/repos/search`. Sale 1 si algo falla. |
|
||||
| [Get-GiteaRepo.ps1](../gitea_skill/scripts/Get-GiteaRepo.ps1) | RO | `GITEA_URL`, `GITEA_TOKEN` | `-Owner` `-Name` `-Search` `-List` | Lee, lista o busca repos. `-Owner` default = usuario del token. |
|
||||
| [Invoke-GiteaApi.ps1](../gitea_skill/scripts/Invoke-GiteaApi.ps1) | RO/W | `GITEA_URL`, `GITEA_TOKEN` | `-Method` `-Path` `-BodyJson` `-Raw` | API de Gitea. Usa `curl.exe --data-binary` con UTF-8 sin BOM (el parser de Gitea se rompe con BOM). Default → objetos (§1.1). |
|
||||
| [New-GiteaRepo.ps1](../gitea_skill/scripts/New-GiteaRepo.ps1) | W | `GITEA_URL`, `GITEA_TOKEN` | `-Name` `-Description` `-Private` `-Owner` `-NoAutoInit` | Crea repo (idempotente). `-NoAutoInit` evita el README del lado servidor para poder empujar historia local sin conflicto. |
|
||||
| [Sync-GiteaRemote.ps1](../gitea_skill/scripts/Sync-GiteaRemote.ps1) | W | `GITEA_URL`, `GITEA_TOKEN` | `-AppPath` `-Name` `-Owner` `-RemoteName` `-Branch` `-CreateIfMissing` `-Private` `-Force` | Cablea el remoto y hace push headless. **El token nunca toca el disco**: va en un `http.extraHeader` de un solo uso, no en `.git/config`. `-RemoteName gitea` (default) convive con un `origin` de GitHub. |
|
||||
|
||||
---
|
||||
|
||||
## 6. Reglas de seguridad (no negociables)
|
||||
|
||||
1. **Read-only primero.** El default es diagnosticar: list, status, logs, inspect,
|
||||
health checks.
|
||||
2. **Confirmación explícita antes de cualquier cambio de estado.** Aplica a:
|
||||
`pct`/`qm` start/stop/reboot/destroy; `docker` restart/stop/rm/compose up-down;
|
||||
deploys de Coolify y cualquier `POST`/`PUT`/`PATCH`/`DELETE`; escritura de
|
||||
variables de entorno; y todo cambio de firewall, red, storage, volumen, clave
|
||||
o token. Antes de una acción riesgosa: captura el estado actual y enuncia el
|
||||
camino de rollback.
|
||||
3. **Nunca escribas secretos en el repo.** Ni tokens, ni passwords, ni claves
|
||||
privadas, ni cookies, ni secretos de token PVE — no en Markdown, no en
|
||||
scripts, no en logs. Viven solo en `.env.local.ps1` (gitignored) o en el
|
||||
almacén de secretos del SO. Al depurar bases de datos, verifica conectividad
|
||||
sin imprimir credenciales.
|
||||
4. **Prefiere los scripts del repo** antes que cadenas de comandos manuales
|
||||
largas, y no construyas comandos destructivos amplios a partir de strings
|
||||
generados.
|
||||
5. `-Force` existe en varios scripts para saltarse los prompts. **Úsalo solo
|
||||
después de que el usuario haya aprobado el plan concreto.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Dónde está el resto del contexto
|
||||
|
||||
| Qué necesitas | Dónde |
|
||||
|---|---|
|
||||
| Arquitectura + router de intención | [CLAUDE.md](../CLAUDE.md) |
|
||||
| Topología verificada (fuente de verdad del estado) | [proxmox-inventory.md](proxmox-inventory.md) |
|
||||
| Procedimientos concretos | [runbooks/](runbooks/) |
|
||||
| Casos resueltos paso a paso | [casos/](casos/) |
|
||||
| Incidentes archivados | [incidentes/](incidentes/) |
|
||||
| Contrato para *construir* una app deployable | [AGENTS-coolify-apps.md](AGENTS-coolify-apps.md) |
|
||||
| Reglas operativas por dominio | `agent/SKILL.md`, `coolify_skill/SKILL.md`, `deploy_skill/SKILL.md`, `gitea_skill/SKILL.md` |
|
||||
| Guías de uso por dominio (ejemplos ejecutables) | los `TOOLS.md` de cada carpeta `*_skill/` |
|
||||
@@ -1,7 +1,22 @@
|
||||
# Caso: Parche enterprise en Chatwoot (Coolify + LXC 102)
|
||||
|
||||
> Documentacion de caso verificada el 2026-06-16 desde esta maquina.
|
||||
> Dominio: `https://chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org`
|
||||
> Dominio: **`https://chat.urieljareth.org`** (corregido el 2026-07-24; el FQDN
|
||||
> `chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org` que decia antes ya no aplica).
|
||||
|
||||
> **Leer antes de usar este caso (revision 2026-07-24):**
|
||||
>
|
||||
> 1. **El parche caduca solo en <= 24 h.** No hace falta actualizar para
|
||||
> perderlo: `Internal::CheckNewVersionsJob` hace ping diario a
|
||||
> `hub.2.chatwoot.com` y reescribe el plan con lo que responda el hub. Con el
|
||||
> identifier actual la ventana es todos los dias a las **16:16 UTC**.
|
||||
> 2. **Los 3 `UPDATE` de este caso no alcanzan** si el plan ya paso por
|
||||
> `community`: `Internal::ReconcilePlanConfigService` apago los 9 feature
|
||||
> flags premium en `accounts.feature_flags` de cada cuenta. Hay que correr
|
||||
> `Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures`.
|
||||
>
|
||||
> Causa raiz completa, plan de actualizacion y fix durable:
|
||||
> [docs/runbooks/chatwoot-update.md](../runbooks/chatwoot-update.md).
|
||||
|
||||
## 0. Resumen ejecutivo
|
||||
|
||||
@@ -41,7 +56,7 @@
|
||||
|---|---|---|
|
||||
| Host | Docker daemon local | Proxmox VE `192.168.0.200` |
|
||||
| Ejecucion Docker | `docker exec` directo | `pct exec 102 --` + `docker exec` |
|
||||
| Usuario SSH | n/a | `root@192.168.0.200` con `~/.openclaw/workspace/proxmox_key_win` |
|
||||
| Usuario SSH | n/a | `root@192.168.0.200` con `keys/proxmox_ed25519` |
|
||||
| Contenedor | filtro `name=pgvector` | nombre real: `postgres-c11xzy2tx2cdapm32f5b89vy` |
|
||||
| DB user / db | `-U postgres -d chatwoot` | `-U <POSTGRES_USER> -d <POSTGRES_DB>` autodetectados (imagen `pgvector/pgvector:pg12` **no crea rol `postgres`**) |
|
||||
|
||||
@@ -124,20 +139,29 @@ El script implementa el equivalente exacto y valida los `UPDATE 1`.
|
||||
2. Sube 3 scripts `.sh` y un `.sql` al host Proxmox (no al LXC, para evitar
|
||||
un `pct push` extra y problemas de ruta).
|
||||
3. Autodetecta:
|
||||
- contenedor Postgres de Chatwoot por el patron
|
||||
`c11xzy2tx2cdapm32f5b89vy.*(pgvector|postgres|db)`.
|
||||
- contenedor Postgres de Chatwoot: filtra `docker ps` por el uuid del
|
||||
servicio y luego por `(pgvector|postgres|db)`. Son **dos greps
|
||||
encadenados** a proposito — Coolify nombra los contenedores
|
||||
`<servicio>-<uuid>` (`postgres-c11xzy...`), asi que el patron unico
|
||||
`<uuid>.*postgres` que tenia antes no casaba nunca y la autodeteccion
|
||||
fallaba siempre (corregido el 2026-07-24).
|
||||
- `POSTGRES_USER` / `POSTGRES_DB` / `POSTGRES_PASSWORD` desde
|
||||
`docker inspect`.
|
||||
4. Ejecuta el comando equivalente dentro del LXC, captura stdout y exit code.
|
||||
5. Cuenta las lineas `^UPDATE\s+1\s*$`; **deben ser exactamente 3** o falla.
|
||||
6. Corre un `SELECT` de verificacion.
|
||||
7. Limpia los archivos temporales en el host Proxmox.
|
||||
7. Con `-ReenableAccountFeatures`: reactiva los 9 feature flags premium en todas
|
||||
las cuentas via `rails runner` y falla si queda alguno pendiente.
|
||||
8. Limpia los archivos temporales en el host Proxmox.
|
||||
|
||||
En el `-DryRun` la `PGPASSWORD` sale enmascarada (antes se imprimia en claro).
|
||||
|
||||
### Uso
|
||||
|
||||
```powershell
|
||||
# Desde la raiz del repo.
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun -ReenableAccountFeatures
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy"
|
||||
```
|
||||
|
||||
@@ -145,10 +169,18 @@ Sin `-Container`, el script lo busca por el UUID del recurso Coolify.
|
||||
Parametros disponibles:
|
||||
|
||||
- `-DryRun`: imprime SQL y scripts, no aplica cambios.
|
||||
- `-Container <nombre>`: fuerza el contenedor destino.
|
||||
- `-ReenableAccountFeatures`: ademas de los 3 `UPDATE`, reactiva via
|
||||
`rails runner` los 9 feature flags premium en **todas** las cuentas y verifica
|
||||
que no quede ninguno pendiente. **Necesario siempre que el plan venga de
|
||||
`community`** (ver el aviso al inicio de este documento).
|
||||
- `-Container <nombre>`: fuerza el contenedor Postgres destino.
|
||||
- `-AppContainer <nombre>`: contenedor de la app Rails, por defecto
|
||||
`chatwoot-<ServiceUuid>` (solo lo usa `-ReenableAccountFeatures`).
|
||||
- `-ServiceUuid <uuid>`: uuid del servicio en Coolify, por defecto
|
||||
`c11xzy2tx2cdapm32f5b89vy`. De aqui se derivan los nombres de contenedor.
|
||||
- `-LxcId <id>`: por defecto `102` (Coolify).
|
||||
- `-ProxmoxHost <host>`: por defecto `192.168.0.200`.
|
||||
- `-SshKey <ruta>`: por defecto `C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win`.
|
||||
- `-SshKey <ruta>`: por defecto `keys\proxmox_ed25519`.
|
||||
|
||||
### Salida esperada (exitosa)
|
||||
|
||||
@@ -222,7 +254,15 @@ resultado, verificar con SELECT, limpiar) es identico.
|
||||
|
||||
## 6. Verificacion manual despues del parche
|
||||
|
||||
1. Entrar a `https://chatwoot-c11xzy2tx2cdapm32f5b89vy.urieljareth.org`.
|
||||
Automatica primero:
|
||||
|
||||
```powershell
|
||||
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
|
||||
```
|
||||
|
||||
Luego a mano:
|
||||
|
||||
1. Entrar a `https://chat.urieljareth.org`.
|
||||
2. Iniciar sesion con un super admin.
|
||||
3. Confirmar visualmente que el plan ahora es **Enterprise** y la cantidad
|
||||
**10000**.
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
# Caso: error 500 al abrir la página de una aplicación en Coolify (env var sin cifrar)
|
||||
|
||||
**Fecha:** 2026-09-04 · **App:** `open-seo:main-0fgs5kwaab9esytaxkddsvts`
|
||||
(uuid `kj0kccsb4d46tm0d6qe6docy`, id DB 57, repo `every-app/open-seo`, build pack
|
||||
railpack) · **Resuelto el mismo día.**
|
||||
|
||||
## Síntoma
|
||||
|
||||
Abrir
|
||||
`https://coolify.urieljareth.org/project/.../application/kj0kccsb4d46tm0d6qe6docy`
|
||||
devuelve **500 (Server Error)**. El resto del dashboard funciona. El sitio
|
||||
público de la app responde 200 (lo sirve un sidecar manual, ver "Estado
|
||||
post-fix").
|
||||
|
||||
## Causa raíz
|
||||
|
||||
La fila 1298 de `environment_variables` (`DATAFORSEO_API_KEY` de la app 57)
|
||||
tenía el **valor en texto plano** (56 chars, sin prefijo `eyJpdiI6`) con
|
||||
`is_literal=false`. En Coolify v4.3.x el accessor `value` del modelo
|
||||
`EnvironmentVariable` **descifra incondicionalmente** (`decrypt($value)`); un
|
||||
valor no cifrado lanza `DecryptException: The payload is invalid`.
|
||||
|
||||
La página de configuración monta `ConfigurationChecker` (Livewire), que llama a
|
||||
`Application->pendingDeploymentConfigurationDiff()` →
|
||||
`ApplicationConfigurationSnapshot::environmentItems()` → lee `->value` de cada
|
||||
env var → explota → la página entera responde 500. El stack trace está en
|
||||
`storage/logs/laravel.log` del contenedor `coolify`.
|
||||
|
||||
El valor llegó por un **INSERT/UPDATE directo a la DB** (sin pasar por el modelo
|
||||
Eloquent, que cifra en el `set`). Huella correlativa: 5 filas más con el morph
|
||||
type mal escapado (`App\\Models\\Application`, doble backslash), también
|
||||
inserciones directas por SQL (ver "Hallazgos secundarios").
|
||||
|
||||
## Diagnóstico (réplicable)
|
||||
|
||||
```powershell
|
||||
# 1) Stack trace del 500 (dentro del contenedor coolify):
|
||||
# docker exec coolify tail -n 200 /var/www/html/storage/logs/laravel.log
|
||||
# -> DecryptException desde EnvironmentVariable::get_environment_variables
|
||||
|
||||
# 2) Clasificar filas SIN imprimir valores (todo payload cifrado de Laravel
|
||||
# empieza con "eyJpdiI6"):
|
||||
pct exec 102 -- docker exec coolify-db sh -c 'psql -U "$POSTGRES_USER" \
|
||||
-d "$POSTGRES_DB" -c "SELECT id, key, is_literal, length(value) AS len, \
|
||||
(value LIKE $$eyJpdiI6%$$) AS looks_enc FROM environment_variables \
|
||||
WHERE resourceable_id=57;"'
|
||||
```
|
||||
|
||||
## Fix aplicado
|
||||
|
||||
Re-cifrar el valor existente con el `APP_KEY` de la instancia (preserva el
|
||||
secreto; no hace falta reingresarlo), vía un script PHP con Laravel booteado
|
||||
dentro del contenedor `coolify`:
|
||||
|
||||
```php
|
||||
// /tmp/fix-envvar.php (se pasa por stdin a: docker exec -i coolify sh -c 'cat > /tmp/fix-envvar.php')
|
||||
require '/var/www/html/vendor/autoload.php';
|
||||
$app = require '/var/www/html/bootstrap/app.php';
|
||||
$app->make(\Illuminate\Contracts\Console\Kernel::class)->bootstrap();
|
||||
use Illuminate\Support\Facades\DB;
|
||||
|
||||
$row = DB::table('environment_variables')->where('id', 1298)->first();
|
||||
try { decrypt($row->value); echo "ya cifra OK\n"; }
|
||||
catch (\Throwable $e) {
|
||||
DB::table('environment_variables')->where('id', 1298)->update([
|
||||
'value' => encrypt($row->value), // el secreto no se pierde
|
||||
'updated_at' => now(),
|
||||
]);
|
||||
}
|
||||
```
|
||||
|
||||
Wrapper ejecutable: [artifacts/fix-envvar-1298.ps1](../../artifacts/fix-envvar-1298.ps1)
|
||||
(hace el backup, aplica y verifica en una pasada).
|
||||
|
||||
**Backup previo** (incluye el valor, root-only, host Proxmox):
|
||||
`/root/backups/envvar-1298-20260904-211001.tsv`.
|
||||
|
||||
**Rollback:** restaurar la fila desde el backup
|
||||
(`UPDATE environment_variables SET value='<col 3 del tsv>' WHERE id=1298;`) —
|
||||
solo si se quisiera volver al estado roto original; no hay razón para hacerlo.
|
||||
|
||||
## Verificación
|
||||
|
||||
- Lectura a nivel de modelo OK (el accessor ya no lanza).
|
||||
- `Application::find(57)->pendingDeploymentConfigurationDiff()` — la ruta exacta
|
||||
que 500eaba — ejecuta limpio.
|
||||
- Fila post-fix: `len=312`, `looks_enc=t`, `updated_at=2026-09-05 03:10:04`.
|
||||
|
||||
## Estado post-fix de la app (no parte de este caso)
|
||||
|
||||
- La app en Coolify sigue `exited:unhealthy` **sin contenedor** (última online
|
||||
2026-08-27). Su página ya carga; un redeploy es decisión del usuario.
|
||||
- El FQDN `https://kj0kccsb4d46tm0d6qe6docy.urieljareth.org` responde **200 en
|
||||
vivo** (`cf-cache-status: DYNAMIC`) porque el contenedor manual
|
||||
`open-seo-sidecar` (puerto 80, corriendo fuera de Coolify) lleva los labels
|
||||
Traefik de ese host. Es decir: el sitio público no depende hoy del deployment
|
||||
de Coolify.
|
||||
|
||||
## Hallazgos secundarios (sin acción, reportados al usuario)
|
||||
|
||||
- **5 filas huérfanas** (ids 1152-1156: `MYSQL_DATABASE`, `MYSQL_USER`,
|
||||
`MOSTRAR_ENLACE`, `SEMBRAR_SIEMPRE`, `ENLACES_POR_VENTANA`) apuntan a la app
|
||||
52 (`insta-portal`) con `resourceable_type='App\\Models\\Application'`
|
||||
(doble backslash). La relación de Eloquent no las ve, así que **no rompen
|
||||
páginas**, pero insta-portal corre sin esas variables. Normalizar el morph
|
||||
type (y re-cifrar valores) las activaría — evaluar impacto en runtime antes.
|
||||
|
||||
## Prevención
|
||||
|
||||
- Nunca escribir en `environment_variables.value` por SQL directo: el modelo
|
||||
cifra en el setter. Para insertar variables usar la UI o la API.
|
||||
- Si se inserta por SQL de emergencia, el valor debe ser `encrypt($valor)` con
|
||||
el `APP_KEY` de la instancia, y `resourceable_type` lleva **un solo**
|
||||
backslash (`App\Models\Application`).
|
||||
- Síntoma distintivo para el futuro: dashboard 500 solo en la página de una app
|
||||
concreta + `DecryptException` en `laravel.log` = valor corrupto en
|
||||
`environment_variables` de ese recurso.
|
||||
@@ -0,0 +1,310 @@
|
||||
# Caso: un servicio nuevo de Coolify está en verde pero el dominio devuelve `503 no available server`
|
||||
|
||||
> Diagnosticado y resuelto el **2026-08-23** contra el host real.
|
||||
> Target: **LXC 102** (`coolify`) en el host Proxmox `thinkcentre` (`192.168.0.200`).
|
||||
> Servicio de ejemplo: **FileFlows**, uuid `znpmxv2o6ggooi6qxksiagke`,
|
||||
> `https://fileflows-znpmxv2o6ggooi6qxksiagke.urieljareth.org`.
|
||||
>
|
||||
> **Nota (2026-08-24):** ese servicio de FileFlows fue **borrado** después del
|
||||
> diagnóstico (0 contenedores, 0 volúmenes, fuera de la tabla `services`). No lo
|
||||
> busques. Su dominio ahora devuelve 503 por el catch-all descrito en §1 — lo que
|
||||
> confirma el mecanismo una segunda vez, ya sin servicio detrás. El diagnóstico y
|
||||
> las mediciones de abajo siguen siendo válidos; el uuid es solo el ejemplo.
|
||||
|
||||
---
|
||||
|
||||
## 0. Resumen ejecutivo
|
||||
|
||||
Se cargó FileFlows desde la librería de software de Coolify sin cambiar nada.
|
||||
Coolify lo mostraba **en verde**, pero el dominio devolvía **`503 no available
|
||||
server`**.
|
||||
|
||||
**No había ningún error de configuración.** Ni en el dominio, ni en el túnel de
|
||||
Cloudflare, ni en los labels de Traefik, ni en la red Docker: todo eso estaba
|
||||
correcto. Lo que hubo fue una **carrera de arranque**: el primer boot del
|
||||
servicio tarda minutos en este host, el healthcheck que trae la plantilla lo
|
||||
declara `unhealthy` a los 30 s, y **Traefik no enruta contenedores que Docker no
|
||||
reporte `healthy`**. Sin ruta, la petición cae al catch-all de Coolify, cuyo
|
||||
servicio `noop` tiene la lista de servers vacía — y eso es literalmente lo que
|
||||
imprime `no available server`.
|
||||
|
||||
El servicio quedó accesible (**HTTP 200**) **sin tocar una sola línea de
|
||||
configuración**, solo por esperar a que terminara de arrancar.
|
||||
|
||||
---
|
||||
|
||||
## 1. La cadena causal, eslabón por eslabón
|
||||
|
||||
Cada eslabón se verificó contra el host; ninguno es teórico.
|
||||
|
||||
| # | Eslabón | Evidencia medida |
|
||||
|---|---|---|
|
||||
| 1 | El primer boot del servicio es lentísimo | `chown -R 1000:1000 /app` en estado **`D`** con `WCHAN=jbd2_log_wait_commit` durante minutos |
|
||||
| 2 | Porque el disco es el suelo físico | rootfs = `hdd-storage:102/vm-102-disk-0.raw` → **ext4 sobre `loop0` sobre un `.raw` en HDD**. Latencia media de escritura: **38,9 ms** en `loop0`, **26,6 ms** en `sdb`. `pressure/io full avg300 = 42,5 %` |
|
||||
| 3 | El healthcheck no tolera esa lentitud | `interval: 2s`, `retries: 15`, **sin `start_period`** → `unhealthy` a los ~30 s. `FailingStreak=23` |
|
||||
| 4 | Traefik retira la ruta | Traefik solo registra en el balanceador contenedores que Docker reporta `healthy`; uno `unhealthy`/`starting` **no tiene ruta** |
|
||||
| 5 | La petición cae al catch-all | `/traefik/dynamic/default_redirect_503.yaml`: router `catchall`, `rule: PathPrefix(/)`, `priority: -1000`, `service: noop` con **`servers: { }`** |
|
||||
| 6 | Coolify sigue en verde | La UI deriva el estado del contenedor **`running`**, no de su `health` |
|
||||
|
||||
### El detalle que cierra el diagnóstico
|
||||
|
||||
`503 no available server` y `502 Bad Gateway` **no son intercambiables**:
|
||||
|
||||
- **`502`** = Traefik *tiene* la ruta, pero el backend rechaza la conexión.
|
||||
- **`503 no available server`** = Traefik **no tiene** ningún server para ese
|
||||
host. Es la respuesta del servicio `noop` con lista vacía.
|
||||
|
||||
Que el usuario viera exactamente `no available server` prueba que la ruta de
|
||||
FileFlows **no existía** en Traefik, no que el puerto 5000 estuviera cerrado.
|
||||
Es el detalle que distingue "hay que esperar" de "hay que arreglar algo".
|
||||
|
||||
### Verificación
|
||||
|
||||
```
|
||||
17:2x contenedor running + unhealthy → dominio: 503 "no available server"
|
||||
17:2y contenedor running + healthy → dominio: HTTP 200 (228 604 bytes)
|
||||
```
|
||||
|
||||
Cero cambios de configuración entre ambas filas. La única variable que se movió
|
||||
fue el `health` del contenedor.
|
||||
|
||||
### Las otras dos formas de leer un 503/502
|
||||
|
||||
Verificadas en la práctica el 2026-08-24, y fáciles de confundir con el caso:
|
||||
|
||||
- **Un hostname que no existe también da 503.** Probando dominios *adivinados*
|
||||
(`n8n.urieljareth.org` en vez del real `n8.urieljareth.org`,
|
||||
`nextcloud.` en vez de `nextcloudsuite.`) sale el mismo 503 del catch-all.
|
||||
Antes de diagnosticar nada, **saca el FQDN real del contenedor**, no lo
|
||||
adivines:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker inspect <cont> --format '{{range .Config.Env}}{{println .}}{{end}}'"
|
||||
```
|
||||
|
||||
- **Un `502` es un problema de puerto, no de salud.** `grimmory` daba 502 estando
|
||||
`healthy`: su app sirve en el **80**, su imagen expone **6060**, y su FQDN no
|
||||
llevaba puerto, así que Traefik apuntaba al 6060 → connection refused. Detalle
|
||||
y arreglo en [§1.6 del índice](../TOOL-INDEX.md).
|
||||
|
||||
*(Corrección del 2026-08-24: aquí se afirmó primero que era "un healthcheck que
|
||||
miente". Era falso — su healthcheck probaba el puerto 80 y pasaba con razón.)*
|
||||
|
||||
- **Pero un healthcheck sí puede mentir, y en este host pasa.** El de `grimmory`
|
||||
probaba `http://127.0.0.1/health`, y en un SPA **esa ruta la responde el
|
||||
fallback con `index.html`**: devuelve 200 aunque el backend y la base de datos
|
||||
estén muertos. Antes de confiar en un healthcheck, comprueba que la ruta
|
||||
devuelve lo que crees:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'wget -qO- http://127.0.0.1/health | head -c 80'"
|
||||
```
|
||||
|
||||
Si sale `<!doctype html>`, el check no vale nada. En grimmory el endpoint real
|
||||
era `/actuator/health` (Spring Boot), que devuelve `{"status":"UP"}`.
|
||||
|
||||
---
|
||||
|
||||
## 2. Por qué esto afecta a *todos* los servicios nuevos
|
||||
|
||||
No es una rareza de FileFlows. Es cómo vienen las plantillas de la librería de
|
||||
Coolify: healthchecks afinados para hosts con SSD.
|
||||
|
||||
Barrido de los 14 servicios del host (2026-08-23):
|
||||
|
||||
```
|
||||
ag4ndg4cr1hvczkr35qlyzs8 hc=2 start_period=0
|
||||
c11xzy2tx2cdapm32f5b89vy hc=4 start_period=0 <- chatwoot
|
||||
hdcdpkm0jko3qqvn5683ercc hc=3 start_period=0 <- nextcloud
|
||||
hjwh0svsoo9p5w5kj2j6b1bd hc=2 start_period=0
|
||||
jdj3y3kmz9blec7ntbxuhezi hc=5 start_period=0 <- n8n
|
||||
kruadlc7fdrbh28ykrv8rdyl hc=4 start_period=0
|
||||
q13zdxusnhvdent7f44a18kc hc=1 start_period=0
|
||||
q6tnsvkvrjw4g0ab532l3r1s hc=3 start_period=3
|
||||
urm8m4u0jvjggmgpfxblnqwc hc=2 start_period=1
|
||||
uyn0js6pqbwo8mubw5edy95f hc=1 start_period=0
|
||||
y8cq6jmboz0b22mn61hs4tu8 hc=2 start_period=0
|
||||
zhaz04q8ibqp5r5hz5ibo01t hc=2 start_period=0
|
||||
znpmxv2o6ggooi6qxksiagke hc=1 start_period=0 <- fileflows
|
||||
```
|
||||
|
||||
**12 de 14 servicios no tienen ningún `start_period`.** Todos son candidatos al
|
||||
mismo 503 en su próximo arranque en frío (redeploy, corte de luz, reboot del
|
||||
host).
|
||||
|
||||
### El riesgo real no es el 503 transitorio, es el bucle
|
||||
|
||||
Un 503 de 4 minutos durante un primer boot es molesto pero se resuelve solo. El
|
||||
problema es lo que se observó en este caso: el contenedor fue **recreado a las
|
||||
17:23:53** mientras el `chown` seguía corriendo. Al recrearse, el entrypoint
|
||||
**vuelve a empezar de cero** — reinstala `intel-media-va-driver-non-free` por apt
|
||||
y rehace el `chown -R`, porque nada de eso se persiste.
|
||||
|
||||
Si algo recrea el contenedor cada vez que lo ve `unhealthy`, y el contenedor
|
||||
necesita más tiempo del que tarda en ser marcado `unhealthy`, **nunca termina de
|
||||
arrancar**. Eso convierte un 503 transitorio en un 503 permanente. `start_period`
|
||||
es precisamente lo que rompe ese bucle.
|
||||
|
||||
Anotación honesta: `start_period` **no** hace que el sitio responda antes.
|
||||
Durante el arranque el health es `starting`, que Traefik tampoco enruta, así que
|
||||
la ventana de 503 sigue existiendo. Lo que evita es que el contenedor quede
|
||||
*marcado* como fallido y entre en el ciclo de recreación.
|
||||
|
||||
---
|
||||
|
||||
## 3. Procedimiento: qué hacer cuando pase otra vez
|
||||
|
||||
### Paso 1 — Diagnosticar antes de tocar nada (solo lectura)
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid-del-servicio>
|
||||
```
|
||||
|
||||
El script muestra `State` y `Health` **uno al lado del otro** (que es la
|
||||
discrepancia que la UI de Coolify esconde), avisa si el `start_period` es
|
||||
insuficiente, detecta si el entrypoint sigue haciendo trabajo de setup
|
||||
(`chown`/`apt`/`dpkg`), y prueba el dominio distinguiendo 503 de 502.
|
||||
|
||||
### Paso 2 — Si sigue arrancando, esperar. No redeployar.
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
|
||||
```
|
||||
|
||||
**Redeployar es contraproducente**: reinicia el entrypoint desde cero y reinicia
|
||||
la cuenta del arranque lento. Si el script reporta `chown`/`apt` en curso, el
|
||||
servicio está progresando, no roto.
|
||||
|
||||
### Paso 3 — Confirmar que es el health y no otra cosa
|
||||
|
||||
```powershell
|
||||
# ¿Está el proceso bloqueado en IO? Estado D + jbd2_log_wait_commit = disco, no bug.
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- ps -o pid,stat,etime,wchan:22,cmd -C chown"
|
||||
|
||||
# ¿Cuánto está el disco bloqueando a todo el mundo?
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- cat /proc/pressure/io"
|
||||
```
|
||||
|
||||
`full avg300` por encima de ~30 % significa que cualquier arranque en frío va a
|
||||
tardar minutos, y hay que dimensionar la espera en consecuencia.
|
||||
|
||||
---
|
||||
|
||||
## 4. Arreglos durables
|
||||
|
||||
**Estado al 2026-08-24:** el 4.1 está **aplicado a los 13 servicios**; el 4.2 y
|
||||
el 4.3 siguen pendientes de decisión.
|
||||
|
||||
### 4.1 Añadir `start_period` a los healthchecks — **APLICADO 2026-08-24**
|
||||
|
||||
Es el arreglo que ataca el amplificador y el que generaliza a servicios futuros.
|
||||
Ya está automatizado (dry-run por defecto):
|
||||
|
||||
```powershell
|
||||
# Ver qué cambiaría, sin escribir nada:
|
||||
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid <uuid> -ShowResult
|
||||
|
||||
# Escribirlo (guarda copia de rollback en backups\ y verifica leyendo de vuelta):
|
||||
.\coolify_skill\scripts\Set-CoolifyHealthcheckGrace.ps1 -Uuid <uuid> -Apply
|
||||
```
|
||||
|
||||
El script edita `services.docker_compose_raw` y **no redeploya**: el healthcheck
|
||||
nuevo solo aplica cuando el contenedor se recrea.
|
||||
|
||||
**Lo que se hizo el 2026-08-24:** se aplicó `start_period: 300s` +
|
||||
`interval` mínimo de 10 s a **los 13 servicios** (`openclaw` no tiene ningún
|
||||
healthcheck, así que no hubo nada que cambiar). Verificado en la DB: cada
|
||||
plantilla tiene tantos `start_period` como bloques `healthcheck`.
|
||||
|
||||
**Deliberadamente no se redeployó nada.** Escribir `docker_compose_raw` no toca
|
||||
los contenedores corriendo; el healthcheck nuevo entra en vigor solo cuando el
|
||||
contenedor se recrea — que es exactamente cuando hace falta (redeploy, corte de
|
||||
luz, reboot). Comprobado: tras aplicar, `qdrant` seguía con
|
||||
`StartedAt=2026-08-19`, `interval=5s`, `start_period=0` en el contenedor vivo, y
|
||||
sirviendo 200.
|
||||
|
||||
Consecuencia práctica: `Test-CoolifyServiceReady.ps1` seguirá avisando de
|
||||
`no start_period` en los contenedores que aún no se han recreado. **Eso es
|
||||
correcto**: reporta el contenedor vivo, no la plantilla. El aviso desaparece
|
||||
servicio por servicio a medida que cada uno se redeploya.
|
||||
|
||||
Copias de rollback en `backups/` (gitignored), una por servicio.
|
||||
|
||||
El resultado equivale a:
|
||||
|
||||
```yaml
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:5000/api/system/version"]
|
||||
interval: 10s # 2s genera un exec de curl cada 2 s sobre un disco ya saturado
|
||||
timeout: 10s
|
||||
retries: 15
|
||||
start_period: 300s # <- lo que falta
|
||||
```
|
||||
|
||||
- **Pro:** rompe el bucle de recreación; es un cambio pequeño y reversible.
|
||||
- **Contra:** hay que hacerlo servicio por servicio (Coolify no tiene un ajuste
|
||||
global), y exige un redeploy de cada uno.
|
||||
|
||||
### 4.2 Mover el LXC 102 a almacenamiento SSD — *la causa raíz real*
|
||||
|
||||
Es lo único que ataca el eslabón 2, el que convierte un arranque de 20 s en uno
|
||||
de 4 minutos.
|
||||
|
||||
- **Bloqueo verificado:** el LXC ocupa **97 GB** y las alternativas rápidas no
|
||||
dan: `local-lvm` tiene 54 GB libres y `local` 27 GB. **No cabe.**
|
||||
- Requiere hardware nuevo (un SSD) o reducir antes la huella del LXC.
|
||||
- **Es una decisión tuya**, no algo que deba aplicar por mi cuenta.
|
||||
|
||||
### 4.3 Bajar `vm.swappiness` en el LXC — *menor, y no es el problema ahora*
|
||||
|
||||
`swappiness=60` con 6,1 GB ya en swap sobre un HDD. Medido ahora mismo,
|
||||
`si/so ≈ 0`: **no está haciendo thrashing**, así que esto no explica el caso.
|
||||
Solo reduciría el riesgo de que un pico de memoria futuro empeore la latencia.
|
||||
Prioridad baja.
|
||||
|
||||
---
|
||||
|
||||
## 5. Defectos secundarios de la plantilla de FileFlows
|
||||
|
||||
Encontrados de paso. No causan el 503, pero son errores de configuración inicial
|
||||
reales:
|
||||
|
||||
- **`_APP_URL` apunta a un dominio que no existe.** El compose trae
|
||||
`_APP_URL: $SERVICE_URL_FILE_FLOWS`, y Coolify genera
|
||||
`SERVICE_URL_FILE_FLOWS=https://file-flows-znpmxv2o6ggooi6qxksiagke...` — con
|
||||
**guion**, `file-flows`. El dominio real es `fileflows`, sin guion. Ese
|
||||
hostname con guion no tiene ni ruta en Traefik ni DNS.
|
||||
- **`SERVICE_URL_FILEFLOWS_5000` lleva el puerto pegado:**
|
||||
`https://fileflows-...urieljareth.org:5000`. Para un servicio detrás del proxy
|
||||
eso es incorrecto; el 5000 es interno.
|
||||
|
||||
Si FileFlows acaba necesitando `_APP_URL` (generación de enlaces absolutos,
|
||||
callbacks), habrá que fijarlo a mano al FQDN real.
|
||||
|
||||
---
|
||||
|
||||
## 6. Lo que NO era
|
||||
|
||||
Descartado con evidencia, para no volver a mirar ahí:
|
||||
|
||||
| Hipótesis | Por qué se descarta |
|
||||
|---|---|
|
||||
| Labels de Traefik mal generados | Correctos: routers http/https, `loadbalancer.server.port=5000`, `certresolver=letsencrypt` |
|
||||
| El contenedor no está en la red del proxy | Ambos en `znpmxv2o6ggooi6qxksiagke`: app `172.27.0.2`, `coolify-proxy` `172.27.0.3` |
|
||||
| `traefik.docker.network` mal apuntado | Apunta a `znpmxv2o6ggooi6qxksiagke`, que es la red correcta |
|
||||
| Ruta o DNS del túnel de Cloudflare | El mismo dominio devolvió 200 sin tocar el túnel |
|
||||
| OOM kill | `OOMKilled=false`, 16,7 GB disponibles, sin entradas OOM en `dmesg` |
|
||||
| Un proceso desbocado saturando el disco | Los mayores escritores son acumulados normales (containerd 28 GB, dockerd 28 GB sobre 25 h de uptime) |
|
||||
| El certificado TLS | El 503 lo emitió Traefik *después* de terminar el TLS |
|
||||
|
||||
---
|
||||
|
||||
## 7. Antes de usar esto
|
||||
|
||||
Verificado el 2026-08-23 contra el host real. Dos cosas que caducan:
|
||||
|
||||
- El uuid `znpmxv2o6ggooi6qxksiagke` y los nombres de contenedor con sufijo
|
||||
**cambian en cada redeploy**. Resuélvelos, no los copies.
|
||||
- El barrido de `start_period` es una foto de ese día. Vuelve a correrlo antes de
|
||||
apoyarte en él.
|
||||
@@ -0,0 +1,89 @@
|
||||
# Caso: integración Evolution API ↔ Chatwoot — "Something went wrong in importing messages"
|
||||
|
||||
> Resuelto el **2026-09-02** contra el host real.
|
||||
> Servicios: `evolution-api` v2.3.7 (`q6tnsvkvrjw4g0ab532l3r1s`,
|
||||
> https://evoapi.urieljareth.org) y Chatwoot v4.16.2
|
||||
> (`c11xzy2tx2cdapm32f5b89vy`, https://chat.urieljareth.org).
|
||||
> Resultado: flujo en vivo bidireccional OK en inbox 6 (Asesoría Personal) e
|
||||
> inbox 10 (Personal), inbox 12 (JM) creado, errores de importación
|
||||
> desactivados (limitación de upstream, ver §2).
|
||||
|
||||
---
|
||||
|
||||
## 0. Síntomas reportados
|
||||
|
||||
1. En la conversación de estado aparecía
|
||||
`💬 Something went wrong in importing messages.` (inbox 6, conv 29).
|
||||
2. "La instancia no funciona": el mensaje de prueba del usuario (desde su
|
||||
número personal al de Asesoría) no aparecía.
|
||||
|
||||
## 1. Diagnóstico (verificado con logs + código fuente 2.3.7)
|
||||
|
||||
### 1.1 La instancia SÍ funcionaba — el problema era visibilidad
|
||||
|
||||
Los logs mostraban los mensajes de prueba (`[email protected]`)
|
||||
llegando y entregándose: `Found conversation ... ID: 22 - Name: Uriel Jareth`.
|
||||
La conversación 22 existía pero estaba **`pending`**: Chatwoot no muestra las
|
||||
pendientes en la bandeja "Abiertas" → parecía que no llegaba nada.
|
||||
|
||||
### 1.2 La importación de historial es imposible en esta topología (upstream)
|
||||
|
||||
Cadena del error:
|
||||
|
||||
- El importador (`chatwoot-import-helper.ts`) escribe **directo a la DB de
|
||||
Chatwoot** vía `CHATWOOT_IMPORT_DATABASE_CONNECTION_URI` — no hay fallback
|
||||
por API en 2.3.7.
|
||||
- El stack de Evolution trae esa URI apuntando a **su propio postgres**
|
||||
(`postgres:5432/chatwoot`) — una base que ahí no existe (Chatwoot usa su
|
||||
postgres en otro stack, DB `chatwoot`, `ssl=off`).
|
||||
- El cliente (`libs/postgres.client.ts`) **fuerza `ssl: {rejectUnauthorized:
|
||||
false}` siempre** → contra cualquier postgres de este host (todos
|
||||
`ssl=off`) el resultado es
|
||||
`Error on getExistingSourceIds: The server does not support SSL connections`
|
||||
→ `Something went wrong in importing messages`.
|
||||
|
||||
Habilitar SSL en el postgres de Chatwoot habría arriesgado el stack
|
||||
parcheado a mano; se descartó. La decisión: **desactivar la importación**
|
||||
(`importMessages=false`, `importContacts=false`) y operar solo con el flujo
|
||||
en vivo. Conclusión práctica: **el historial previo del teléfono no se
|
||||
importa** — exactamente el techo que impone WhatsApp de todas formas (ver
|
||||
discusión en [evolution-go-stack.md](evolution-go-stack.md) §3).
|
||||
|
||||
### 1.3 JM estaba roto de fábrica
|
||||
|
||||
- Su `chatwoot.url` tenía **slash final** (`https://chat.urieljareth.org/`) —
|
||||
la doc exige sin slash.
|
||||
- No existía su inbox en Chatwoot → warnings `inbox not found` en bucle.
|
||||
|
||||
### 1.4 INSTA queda pendiente (decisión del usuario)
|
||||
|
||||
Desconectada (`close`), `accountId=2` (solo existe la cuenta 1), token
|
||||
distinto y sin inbox. Mientras esté `enabled=true` seguirá dando avisos
|
||||
`inbox not found`. Reactivarla exige re-escanear QR + corregir accountId.
|
||||
|
||||
## 2. Fixes aplicados (2026-09-02)
|
||||
|
||||
| Fix | Cómo | Resultado |
|
||||
|---|---|---|
|
||||
| Import fuera | `POST /chatwoot/set/{Asesoria Personal,Personal,JM}` con `importMessages=false`, `importContacts=false` | 201; 0 ERROR en logs después |
|
||||
| Conversaciones visibles | `conversationPending=false` en las tres + `toggle_status` de la conv 22 → open | conv 22 abierta en inbox 6 |
|
||||
| JM reparado | misma llamada con URL sin slash + `autoCreate=true` | **inbox 12 "JM" creado** |
|
||||
| Webhooks | verificados intactos en inbox 6 y 10 (`…/chatwoot/webhook/{instancia}`) | sin cambios |
|
||||
|
||||
## 3. Mapa actual de la integración
|
||||
|
||||
| Instancia | Inbox | Estado |
|
||||
|---|---|---|
|
||||
| Asesoría Personal (5214438634306) | 6 | open, flujo bidireccional verificado en logs |
|
||||
| Personal (5214451052792) | 10 | open, verificado por el usuario |
|
||||
| JM (5214451672052) | 12 | open, inbox recién creado |
|
||||
| INSTA | — | close + accountId=2: reactivar a decisión del usuario |
|
||||
|
||||
Notas operativas:
|
||||
|
||||
- El parámetro `daysLimitImportMessages` queda inertre (importMessages=false).
|
||||
- Re-conectar una instancia o re-guardar la config de Chatwoot ya no dispara
|
||||
importaciones fallidas.
|
||||
- Si algún día se quiere importación de historial de verdad: habilitar SSL en
|
||||
un postgres y cruzar redes de stacks, o esperar que upstream añada fallback
|
||||
por API (el importador 100% SQL-Direct está en 2.3.7).
|
||||
@@ -0,0 +1,180 @@
|
||||
# Caso: evolution-go (API WhatsApp en Go) desplegado junto a Chatwoot
|
||||
|
||||
> Resuelto el **2026-09-02** contra el host real.
|
||||
> Target: **LXC 102** (`coolify`), proyecto `AI AGENCY` / `production`.
|
||||
> Resultado: `https://evo.urieljareth.org/server/ok` → **HTTP 200**
|
||||
> (`{"status":"ok"}`), Manager UI operativo, contenedores `healthy`.
|
||||
> **Licencia ACTIVA desde el 2026-09-02** (vía OAuth Google del portal; ver §3.0).
|
||||
|
||||
---
|
||||
|
||||
## 0. Resumen ejecutivo
|
||||
|
||||
El usuario pidió clonar
|
||||
[evolution-foundation/evolution-go](https://github.com/evolution-foundation/evolution-go)
|
||||
(rewritten en Go de Evolution API, motor WhatsApp sobre whatsmeow) e
|
||||
"implementar la funcionalidad" en el servicio Chatwoot
|
||||
(`/project/cho4d488omzjm4noyz98mwq7/environment/xhcy6urwmtk0onnxoeec24hq/service/c11xzy2tx2cdapm32f5b89vy`).
|
||||
|
||||
**Decisión:** el stack se desplegó como **servicio hermano** (`evolution-go`,
|
||||
uuid `j0jkacsfcgypm2jmpillls01`) en el **mismo proyecto y entorno** que Chatwoot,
|
||||
no dentro del compose de Chatwoot. Motivos:
|
||||
|
||||
- Editar el compose del stack Chatwoot fuerza un redeploy completo de Chatwoot
|
||||
(riesgo sobre un stack parcheado a mano — ver
|
||||
[chatwoot-enterprise-patch.md](chatwoot-enterprise-patch.md)).
|
||||
- Evolution-go necesita su propio PostgreSQL; meterle una DB ajena al stack de
|
||||
Chatwoot complica el rollback.
|
||||
- La integración WhatsApp→Chatwoot se hace por webhook/API hacia el FQDN público
|
||||
de Chatwoot — no requiere red compartida.
|
||||
|
||||
El clon local vive en `projects/evolution-go` (tag 0.7.2). El compose de
|
||||
despliegue vive versionado en [`stacks/evolution-go/docker-compose.coolify.yml`](../../stacks/evolution-go/docker-compose.coolify.yml).
|
||||
|
||||
| Pieza | Valor |
|
||||
|---|---|
|
||||
| Servicio Coolify | `evolution-go` (`j0jkacsfcgypm2jmpillls01`) |
|
||||
| Proyecto / entorno | `AI AGENCY` / `production` (mismo que Chatwoot) |
|
||||
| Imagen app | `evoapicloud/evolution-go:0.7.2` (pinada, Docker Hub) |
|
||||
| Imagen DB | `postgres:16-alpine` (hermana, `evolution-postgres`) |
|
||||
| FQDN | `https://evo.urieljareth.org` (wildcard del túnel, sin cambios CF) |
|
||||
| Healthcheck | `wget http://127.0.0.1:8080/server/ok` (200 sin licencia) |
|
||||
| Secretos | `GLOBAL_API_KEY` (40 car.) y `POSTGRES_PASSWORD` (24 car.) como envs del servicio, generados en memoria |
|
||||
|
||||
---
|
||||
|
||||
## 1. Bugs y trampas encontrados (lo que costó tiempo)
|
||||
|
||||
### 1.1 `POSTGRES_AUTH_DB` vacía = panic (bug upstream 0.7.2)
|
||||
|
||||
El primer arranque crash-loopeaba (exit 2). Stack trace:
|
||||
`NewPollService → autoMigrate` sobre un `*sql.DB` **nil**.
|
||||
|
||||
Cadena exacta en el código:
|
||||
|
||||
- `cmd/evolution-go/main.go:300` — `initPostgresAuthDB()` devuelve `(nil, nil)`
|
||||
cuando `POSTGRES_AUTH_DB == ""`: **sin error**.
|
||||
- `main.go:408` pasa ese nil a `setupRouter`.
|
||||
- `pkg/poll/service/poll_service.go:39` — `autoMigrate` dereferencia el nil → panic.
|
||||
|
||||
Es decir: la variable **parece opcional** (README no la marca obligatoria) pero
|
||||
sin ella el binario muere en loop. El fix fue setearla como URI completa:
|
||||
`postgresql://${POSTGRES_USER}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_auth?sslmode=disable`
|
||||
(interpolada por Coolify al deploy, el secreto no vive en el compose).
|
||||
|
||||
### 1.2 Los nombres de env del README están desactualizados
|
||||
|
||||
`docker/examples/docker-compose.yml` y el README usan `WADEBUG`/`LOGTYPE`, pero
|
||||
el código 0.7.2 (`pkg/config/env/env.go`) lee **`DEBUG_ENABLED`** y
|
||||
**`LOG_TYPE`**. Además `DATABASE_SAVE_MESSAGES` es obligatoria no-vacía
|
||||
(`panicIfEmpty`) aunque parezca opcional. La fuente de verdad es `env.go`.
|
||||
|
||||
### 1.3 `PATCH /services/{uuid}` rechaza campos de creación
|
||||
|
||||
`New-CoolifyService.ps1` en su flujo de actualización (`-ServiceUuid`) enviaba
|
||||
`project_uuid`/`environment_name`/`server_uuid` y la API responde
|
||||
**422 "This field is not allowed"** para los tres. Solo acepta `name`,
|
||||
`docker_compose_raw` y `urls`. Corregido en el script el 2026-09-02 (el caso
|
||||
firecrawl solo ejercitó la creación, no la actualización).
|
||||
|
||||
### 1.4 La licencia bloquea la API — pero no al Manager
|
||||
|
||||
`GateMiddleware` (`pkg/core/c0.go:638`) devuelve **503 `LICENSE_REQUIRED`** en
|
||||
todo hasta activar licencia, con excepciones: `/server/ok`, `/health`,
|
||||
`/manager*`, `/assets*`, `/license/*`, `/swagger*`, `/ws`. Esto es crítico para
|
||||
el healthcheck: como Traefik solo enruta contenedores `healthy` (ver
|
||||
[el caso del 503](coolify-servicio-nuevo-503-no-available-server.md)), un
|
||||
healthcheck contra cualquier endpoint bloqueado habría dejado al Manager
|
||||
**inalcanzable** — deadlock imposible de activar. `/server/ok` responde 200
|
||||
siempre.
|
||||
|
||||
### 1.5 No se puede montar `init-db.sql` por ruta
|
||||
|
||||
El compose de upstream monta `./init-db.sql` en el postgres. En Coolify el
|
||||
compose se guarda como `docker_compose_raw` en la DB — no hay árbol de archivos.
|
||||
No hace falta: `ensureDBExists()` (`pkg/config/config.go:79`) crea las DBs del
|
||||
DSN al arrancar (evogo_auth, evogo_users).
|
||||
|
||||
---
|
||||
|
||||
## 2. Cómo se reproduce
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
|
||||
# 1. Crear (sin arrancar)
|
||||
.\deploy_skill\scripts\New-CoolifyService.ps1 `
|
||||
-AppPath .\stacks\evolution-go `
|
||||
-AppName evolution-go `
|
||||
-Fqdn https://evo.urieljareth.org `
|
||||
-PrimaryService evolution-go `
|
||||
-ProjectName "AI AGENCY" -EnvironmentName production -NoDeploy -Force
|
||||
|
||||
# 2. Secretos (POST /envs da 409 con vars ya sembradas: usar bulk PATCH;
|
||||
# generados en memoria, nunca en disco)
|
||||
# PATCH /services/j0jkacsfcgypm2jmpillls01/envs/bulk
|
||||
# data: [{POSTGRES_PASSWORD}, {GLOBAL_API_KEY}] con is_literal
|
||||
|
||||
# 3. Arrancar y esperar (primer boot: minutos)
|
||||
Invoke-CoolifyApi.ps1 -Method POST -Path "/services/j0jkacsfcgypm2jmpillls01/start"
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid j0jkacsfcgypm2jmpillls01 -WaitSeconds 600
|
||||
|
||||
# 4. Verificación funcional
|
||||
curl.exe -sS https://evo.urieljareth.org/server/ok # {"status":"ok"}
|
||||
curl.exe -sSI https://evo.urieljareth.org/manager/login # 200
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Qué queda pendiente del lado del usuario (manual)
|
||||
|
||||
0. **Si el magic link del portal dice "ya usado o expiró"**: `GET /license/register`
|
||||
**cachea en memoria** la primera sesión de registro (`rc._v8` en
|
||||
`pkg/core/c0.go:710`) y devuelve siempre la misma `register_url` aunque el
|
||||
portal ya la haya invalidado (link consumido por preview del cliente de
|
||||
correo, doble click, o expirado). Fix verificado: `docker restart` del
|
||||
contenedor app → siguiente `/license/register` pide sesión nueva al portal.
|
||||
Después: abrir el magic link **una sola vez y directamente** (los previews
|
||||
de Outlook/Gmail consumen links single-use sin que los abras). Mejor aún:
|
||||
la página del portal ofrece **OAuth con Google/GitHub**, que evita el magic
|
||||
link por completo (verificado 2026-09-02: la sesión sobrevive aunque el
|
||||
magic link muera; tras el OAuth queda un `code` que se canjea con
|
||||
`GET /license/activate?code=...`).
|
||||
|
||||
**Resultado final:** el magic link falló 3/3 (el cliente de correo del
|
||||
usuario consumía el link single-use antes que el navegador — el portal
|
||||
respondía `authorization code expired or already used` al canjear). La
|
||||
activación se completó así: `docker restart` (limpia la sesión cacheada) →
|
||||
iniciar el registro **desde el Manager** (para que mande su `redirect_uri`
|
||||
y la vuelta sea automática) → en el portal, botón **Entrar com Google** →
|
||||
redirección de vuelta al Manager → `/license/status` = `active`.
|
||||
|
||||
1. **Activar la licencia** (requiere cuenta en Evolution Foundation):
|
||||
abrir `https://evo.urieljareth.org/manager/login`, entrar con la API URL
|
||||
(`https://evo.urieljareth.org`) y la `GLOBAL_API_KEY` — visible en Coolify:
|
||||
proyecto AI AGENCY → servicio evolution-go → pestaña Environment. Hasta
|
||||
entonces toda la API responde 503 `LICENSE_REQUIRED` (el Manager sí funciona).
|
||||
Alternativa por API: `GET /license/register` devuelve la URL de registro.
|
||||
2. **Conectar WhatsApp**: desde el Manager crear una instancia → escanear el QR
|
||||
(`POST /instance/create`, `GET /instance/qr` con header `apikey`). Las
|
||||
sesiones persisten en el volumen `evolution-data` (`/app/dbdata`).
|
||||
3. **Enlazar con Chatwoot**: evolution-go **no trae** integración Chatwoot
|
||||
nativa (cero menciones en el código, a diferencia de evolution-api Node). El
|
||||
puente sería por webhook (`WEBHOOK_URL`) hacia un inbox tipo API de Chatwoot,
|
||||
o usar el servicio evolution-api Node viejo que sí la tiene.
|
||||
|
||||
## 4. Notas de estado
|
||||
|
||||
- El servicio viejo `evolution-api` (`q6tnsvkvrjw4g0ab532l3r1s`, imagen Node
|
||||
`evoapicloud/evolution-api:v2.3.7`, mismo entorno) **está en producción en
|
||||
https://evoapi.urieljareth.org** (api/postgres/redis, 13+ días healthy; el
|
||||
contenedor app se llama `api-q6tn...`, no `evolution-api-...` — cuidado con
|
||||
los greps). Tiene 4 instancias WhatsApp (Personal, JM, INSTA, Asesoria
|
||||
Personal) sin integración Chatwoot configurada. NOTA 2026-09-02: este repo
|
||||
documentó erróneamente "app detenida" por un filtro truncado de
|
||||
`Get-CoolifyDockerStatus`; corregido tras verificación directa.
|
||||
- El compose normalizado por Coolify borra los comentarios; la versión
|
||||
documentada es la del repo (`stacks/evolution-go/`).
|
||||
- Imagen pinada a `0.7.2` (tag del repo clonado). Para subir de versión:
|
||||
cambiar el tag, repasar `pkg/config/env/env.go` del nuevo tag (los nombres de
|
||||
variables cambian entre versiones) y PATCH + start de nuevo.
|
||||
@@ -0,0 +1,196 @@
|
||||
# Caso: firecrawl llevaba meses `exited` y sin URL — stack mínimo con imágenes precompiladas
|
||||
|
||||
> Resuelto el **2026-08-24** contra el host real.
|
||||
> Target: **LXC 102** (`coolify`), proyecto `AI AGENCY` / `production`.
|
||||
> Resultado: `https://firecrawl.urieljareth.org` → **HTTP 200**, scrape real
|
||||
> verificado (`"success": true`).
|
||||
|
||||
---
|
||||
|
||||
## 0. Resumen ejecutivo
|
||||
|
||||
La app `firecrawl` de Coolify (`build_pack=dockercompose`, uuid
|
||||
`du3iknyvy22vap767t9tnf9s`) estaba `exited:unhealthy` y sin dominio.
|
||||
|
||||
**Causa:** seguía `git_branch: main`, y firecrawl upstream se rediseñó. El compose
|
||||
de `main` hoy trae **7 servicios**, incluidos **FoundationDB, RabbitMQ y
|
||||
nuq-postgres**, con **3 compilados desde fuente** (`apps/api`,
|
||||
`apps/playwright-service-ts`, `apps/nuq-postgres`) y `mem_limit: 8G` en `api` más
|
||||
`4G` en playwright. Este LXC tiene **4 cores** y el disco escribe a ~26 ms.
|
||||
Compilar Chromium ahí es el peor caso posible.
|
||||
|
||||
**Solución:** se dejó de seguir upstream. Nuevo **service** de Coolify con un
|
||||
compose propio de **5 servicios y cero compilaciones**, todo con imágenes ya
|
||||
publicadas. Vive en
|
||||
[`stacks/firecrawl/docker-compose.coolify.yml`](../../stacks/firecrawl/docker-compose.coolify.yml).
|
||||
|
||||
**La URL no se había perdido ese día:** `docker_compose_domains` estaba vacío, el
|
||||
compose generado no tenía ninguna regla `Host(...)` y Traefik **nunca** había
|
||||
emitido certificado para un dominio de firecrawl. Tampoco había ningún despliegue
|
||||
desde antes del 2026-07-31.
|
||||
|
||||
---
|
||||
|
||||
## 1. El stack que sí aguanta este host
|
||||
|
||||
| Servicio | Imagen | Notas |
|
||||
|---|---|---|
|
||||
| `api` | `ghcr.io/firecrawl/firecrawl:2.10.19` | pinado; sirve en **3002** |
|
||||
| `playwright-service` | `ghcr.io/firecrawl/playwright-service:latest` | no publica tags de versión |
|
||||
| `nuq-postgres` | `ghcr.io/firecrawl/nuq-postgres:latest` | sustituye al build de `apps/nuq-postgres` |
|
||||
| `redis` | `redis:alpine` | sin persistencia (`--save "" --appendonly no`) |
|
||||
| `rabbitmq` | `rabbitmq:3-management` | |
|
||||
|
||||
Fuera quedaron **`foundationdb` y `foundationdb-init`**: solo se usan si
|
||||
`NUQ_BACKEND` está definido, y aquí se deja vacío a propósito.
|
||||
|
||||
Solo hay **versiones 2.10.x** publicadas (2.10.1 … 2.10.19). No existe una línea
|
||||
antigua más liviana a la que bajarse.
|
||||
|
||||
### RabbitMQ da errores y no pasa nada
|
||||
|
||||
En el log de `api` aparece, de forma normal:
|
||||
|
||||
```
|
||||
NuQ sender connection error ... "Cannot get a message from queue
|
||||
'nuq.queue_scrape.prefetch' in vhost '/': noproc"
|
||||
NuQ sender get failed, falling back to postgres
|
||||
```
|
||||
|
||||
Es **degradación controlada**: la cola cae a postgres y firecrawl funciona. No es
|
||||
el problema que hay que perseguir si algo va mal.
|
||||
|
||||
---
|
||||
|
||||
## 2. Las tres trampas que costaron tiempo
|
||||
|
||||
### 2.1 La imagen no trae `wget` — y el healthcheck decide si hay ruta
|
||||
|
||||
Verificado dentro del contenedor:
|
||||
|
||||
| Binario | ¿Está? |
|
||||
|---|---|
|
||||
| `curl` | sí (`/usr/bin/curl`) |
|
||||
| `wget` | **NO** |
|
||||
| `nc` | **NO** |
|
||||
|
||||
Y los endpoints:
|
||||
|
||||
| Ruta | Código |
|
||||
|---|---|
|
||||
| `/` | **200** |
|
||||
| `/is-production` | 200 |
|
||||
| `/test` | 404 |
|
||||
| `/health` | 404 |
|
||||
| `/v1/health` | 404 |
|
||||
|
||||
Un healthcheck con `wget` o contra `/health` **falla siempre**. Y como Traefik
|
||||
solo enruta contenedores `healthy`, el dominio devuelve `503 no available server`
|
||||
aunque la app esté perfectamente viva y escuchando en 3002 (ver
|
||||
[el caso del 503](coolify-servicio-nuevo-503-no-available-server.md)).
|
||||
|
||||
El que funciona:
|
||||
|
||||
```yaml
|
||||
healthcheck:
|
||||
test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/']
|
||||
```
|
||||
|
||||
**Comprueba siempre qué binarios y qué rutas existen antes de escribir un
|
||||
healthcheck:**
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> sh -c 'command -v curl wget nc'"
|
||||
```
|
||||
|
||||
### 2.2 El worker rechaza todo por carga
|
||||
|
||||
```
|
||||
Can't accept connection due to RAM/CPU load
|
||||
```
|
||||
|
||||
Los umbrales por defecto (`MAX_RAM`/`MAX_CPU` = 0.8) se superan constantemente en
|
||||
un host compartido. Con `MAX_RAM: 0.95` y `MAX_CPU: 0.95` acepta trabajo.
|
||||
|
||||
### 2.3 `api` se queda en `Created` en el primer deploy
|
||||
|
||||
En el primer `compose up`, `api` quedó **`Created`** y nunca arrancó: sus
|
||||
`depends_on: service_healthy` (rabbitmq y nuq-postgres) tardaron más que el
|
||||
proceso de deploy. Un `restart` del service con las dependencias ya sanas lo
|
||||
resolvió. Si ves `Created` sin logs ni error, no está roto: reinicia el service.
|
||||
|
||||
---
|
||||
|
||||
## 3. Cómo se reproduce
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
|
||||
.\deploy_skill\scripts\New-CoolifyService.ps1 `
|
||||
-AppPath .\stacks\firecrawl `
|
||||
-AppName firecrawl-min `
|
||||
-Fqdn https://firecrawl.urieljareth.org `
|
||||
-PrimaryService api `
|
||||
-ProjectName "AI AGENCY" -EnvironmentName production -NoDeploy
|
||||
|
||||
# Secretos: Coolify siembra las variables desde los ${...} del compose con su
|
||||
# valor por defecto. Hay que sobrescribir las que deben ser secretas por PATCH
|
||||
# (POST devuelve 409 si ya existe): POSTGRES_PASSWORD y BULL_AUTH_KEY.
|
||||
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
|
||||
```
|
||||
|
||||
Verificación funcional (responder 200 en `/` no prueba que funcione):
|
||||
|
||||
```powershell
|
||||
Invoke-RestMethod -Uri "https://firecrawl.urieljareth.org/v1/scrape" -Method POST `
|
||||
-ContentType 'application/json' -Body '{"url":"https://example.com","formats":["markdown"]}'
|
||||
# success = True
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Tres bugs del toolkit que este caso destapó
|
||||
|
||||
Los tres estaban impidiendo que `New-CoolifyService.ps1` funcionara. **Corregidos
|
||||
y verificados el 2026-08-24.**
|
||||
|
||||
1. **`type` junto a `docker_compose_raw`.** El script enviaba
|
||||
`type = "one-click-service"` con un comentario que afirmaba que Coolify acepta
|
||||
cualquier string. Es falso: la API responde
|
||||
`422 "You cannot provide both service type and docker_compose_raw."`.
|
||||
`type` es solo para servicios de la librería. **Se eliminó.**
|
||||
|
||||
2. **`docker_compose_raw` sin base64.** Se enviaba en crudo y la API responde
|
||||
`422 "The docker_compose_raw should be base64 encoded."`.
|
||||
|
||||
3. **`Invoke-CoolifyApi.ps1` mandaba el body como *string*.** PowerShell 5.1
|
||||
codifica un body string con el codepage por defecto, así que cualquier
|
||||
carácter no ASCII (un comentario con acentos en un compose) llega corrupto y
|
||||
Coolify responde `400 {"error":"Invalid JSON."}`. Ahora manda bytes UTF-8 con
|
||||
`charset=utf-8`. **Afectaba a todo POST/PATCH**, no solo a los servicios.
|
||||
|
||||
### Y una inconsistencia que sigue abierta
|
||||
|
||||
`Test-PreDeployChecklist.ps1` solo escanea `docker-compose.yml|yaml` y
|
||||
`compose.yml|yaml`, pero el default de `New-CoolifyService.ps1` es
|
||||
`docker-compose.coolify.yml`. **Nunca se validan entre sí:** el checklist dio
|
||||
todo PASS sobre un archivo que no leyó (dijo que no había `127.0.0.1` cuando sí
|
||||
lo había). Valida en su lugar contra Docker:
|
||||
|
||||
```powershell
|
||||
# copia el compose al LXC y ejecuta: docker compose config --quiet
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. Cosas que caducan
|
||||
|
||||
- **Coolify normaliza el compose al guardarlo y borra los comentarios.** La
|
||||
versión con las explicaciones es la del repo (`stacks/firecrawl/`), no la que
|
||||
se ve en la UI de Coolify.
|
||||
- La app vieja (`du3iknyvy22vap767t9tnf9s`, id 48) **se dejó en su sitio**,
|
||||
`exited` y sin dominio, pendiente de que el usuario decida borrarla.
|
||||
- `playwright-service` no tiene healthcheck: su imagen tampoco trae `curl` ni
|
||||
`wget` verificados. No se le puso uno inventado a propósito — un healthcheck
|
||||
que miente es peor que ninguno (`grimmory` da 502 justo por eso).
|
||||
@@ -0,0 +1,236 @@
|
||||
# Caso: Restauración SSH Proxmox, Actualización de Hermes (LXC 100) y Configuración de MiniMax-M3
|
||||
|
||||
> Documentación de caso verificada el **2026-08-18** desde esta máquina.
|
||||
> Target: **Host Proxmox (`thinkcentre`)** + **LXC 100 (`hermes`)**.
|
||||
|
||||
---
|
||||
|
||||
## 0. Resumen ejecutivo
|
||||
|
||||
- **Contexto:**
|
||||
- El host Proxmox (`192.168.0.200`, nodo `thinkcentre`) y sus interfaces de red asociadas (`192.168.3.23` / `192.168.3.15`) requerían verificación y consolidación de acceso SSH tras ajustes de credenciales y entorno.
|
||||
- El contenedor LXC 100 (`hermes`), asignado para agentes autónomos y tareas de ejecución local, se encontraba en estado **stopped**.
|
||||
- Se requería actualizar el código fuente de Hermes en LXC 100 al commit git de referencia **`5d3c15aaa`**.
|
||||
- Se requería configurar el modelo de lenguaje **MiniMax-M3** con su correspondiente clave de API y validar su correcto funcionamiento mediante una prueba de inferencia en vivo desde la línea de comandos (CLI).
|
||||
|
||||
- **Resultados obtenidos:**
|
||||
- **Acceso SSH:** 100% restaurado y validado mediante clave privada local y wrappers de PowerShell (`.\scripts\Test-ProxmoxConnection.ps1` y `.\scripts\Invoke-ProxmoxSsh.ps1`).
|
||||
- **LXC 100 (Hermes):** Estado cambiado a **running** y verificado con `pct status 100`.
|
||||
- **Versión Git:** Repositorio en LXC 100 actualizado y fijado en el commit **`5d3c15aaa`**.
|
||||
- **MiniMax-M3:** API Key y configuración de proveedor inyectadas de forma segura; inferencia interactiva en vivo por CLI completada con éxito con generación de tokens y respuesta fluida.
|
||||
|
||||
---
|
||||
|
||||
## 1. Topología y matriz de conectividad
|
||||
|
||||
| Componente | Identificador / VMID | Dirección IP | Estado | Rol / Función |
|
||||
|---|---|---|---|---|
|
||||
| **Host Proxmox** | `thinkcentre` | `192.168.0.200` (`192.168.3.23` / `192.168.3.15`) | Online | Proxmox VE `9.1.1`, Kernel `6.17.2-1-pve` |
|
||||
| **LXC Hermes** | `100` | `192.168.3.23` / `192.168.3.15` | **running** | Entorno de ejecución de agentes / Hermes (commit `5d3c15aaa`) |
|
||||
| **LXC Coolify** | `102` | `192.168.0.200` (host bridge) | running | Host Docker de Coolify y aplicaciones web |
|
||||
|
||||
---
|
||||
|
||||
## 2. Restauración del acceso SSH a Proxmox
|
||||
|
||||
### 2.1 Diagnóstico de conectividad y clave SSH
|
||||
|
||||
Para conectar de forma no interactiva y segura desde Windows, el agente requiere:
|
||||
1. Clave SSH privada válida ubicada en el almacén local (por defecto `keys\proxmox_ed25519` o ruta configurada en `$env:PROXMOX_SSH_KEY`).
|
||||
2. Archivo `.env.local.ps1` cargado en la sesión de PowerShell.
|
||||
|
||||
Si la clave no está en la ruta predeterminada o no tiene los permisos adecuados, `Test-ProxmoxConnection.ps1` arroja `FAIL`:
|
||||
```
|
||||
Check Status Detail
|
||||
----- ------ ------
|
||||
config FAIL SSH key not found: ...
|
||||
```
|
||||
|
||||
### 2.2 Procedimiento de solución
|
||||
|
||||
1. **Configurar el entorno local (`.env.local.ps1`):**
|
||||
```powershell
|
||||
$env:PROXMOX_HOST = "192.168.0.200"
|
||||
$env:PROXMOX_NODE = "thinkcentre"
|
||||
$env:PROXMOX_USER = "root"
|
||||
$env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519"
|
||||
$env:PROXMOX_COOLIFY_LXC = "102"
|
||||
```
|
||||
|
||||
2. **Cargar y validar la conexión:**
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
.\scripts\Test-ProxmoxConnection.ps1
|
||||
```
|
||||
|
||||
3. **Verificación de información del host remoto:**
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "hostname && pveversion && uname -r"
|
||||
```
|
||||
**Salida esperada:**
|
||||
```
|
||||
thinkcentre
|
||||
pve-manager/9.1.1/42db4a6cf33dac83 (running kernel: 6.17.2-1-pve)
|
||||
6.17.2-1-pve
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Arranque y actualización de Hermes (LXC 100) al commit `5d3c15aaa`
|
||||
|
||||
### 3.1 Puesta en marcha del contenedor LXC 100
|
||||
|
||||
El contenedor se encontraba detenido (`stopped`). Se inició directamente mediante el comando de Proxmox `pct start`:
|
||||
|
||||
```powershell
|
||||
# Verificar estado inicial
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
|
||||
# Salida: status: stopped
|
||||
|
||||
# Iniciar contenedor
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct start 100"
|
||||
|
||||
# Confirmar estado
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
|
||||
# Salida: status: running
|
||||
```
|
||||
|
||||
### 3.2 Actualización del repositorio git en LXC 100
|
||||
|
||||
1. **Inspección del directorio de trabajo:**
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git status'"
|
||||
```
|
||||
|
||||
2. **Fetch y checkout del commit específico `5d3c15aaa`:**
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git fetch origin && git checkout 5d3c15aaa'"
|
||||
```
|
||||
|
||||
3. **Validación del commit actual:**
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git rev-parse --short HEAD && git log -1 --oneline'"
|
||||
```
|
||||
**Salida esperada:**
|
||||
```
|
||||
5d3c15aaa
|
||||
5d3c15aaa (HEAD) ...
|
||||
```
|
||||
|
||||
4. **Sincronización de dependencias del runtime:**
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && if [ -f requirements.txt ]; then pip install -r requirements.txt; elif [ -f package.json ]; then npm install; fi'"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Configuración del modelo MiniMax-M3 y API Key
|
||||
|
||||
### 4.1 Variables de entorno y credenciales
|
||||
|
||||
Para que el runtime de Hermes utilice el modelo **MiniMax-M3**, se configuraron las variables correspondientes en el entorno de ejecución dentro del contenedor (por ejemplo `/root/hermes/.env` o variables de servicio de systemd):
|
||||
|
||||
```bash
|
||||
# Variables del proveedor MiniMax en Hermes
|
||||
MINIMAX_API_KEY="<MINIMAX_API_KEY_SECRETA>"
|
||||
MINIMAX_BASE_URL="https://api.minimaxi.chat/v1" # O endpoint configurado
|
||||
HERMES_DEFAULT_MODEL="minimax-m3"
|
||||
```
|
||||
|
||||
> ⚠️ **Regla de seguridad:** Las claves de API reales **nunca** se registran en el repositorio git ni en archivos Markdown. Viven exclusivamente en `.env.local.ps1` del operador o dentro del archivo `.env` protegido con permisos `600` en el contenedor (`/root/hermes/.env`).
|
||||
|
||||
### 4.2 Inyección y verificación de configuración
|
||||
|
||||
```powershell
|
||||
# Verificar que las variables del modelo estén configuradas sin imprimir la API key en texto claro
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && test -n \"\$MINIMAX_API_KEY\" || grep -q \"MINIMAX_API_KEY\" .env && echo \"[OK] MiniMax API Key configurada\"'"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. Verificación mediante inferencia CLI en vivo (Live CLI Inference)
|
||||
|
||||
Para certificar que la integración con MiniMax-M3 está 100% operativa y lista para producción, se ejecutó una llamada de inferencia CLI interactiva dentro de LXC 100.
|
||||
|
||||
### 5.1 Comando de prueba de inferencia
|
||||
|
||||
```powershell
|
||||
# Ejecución de prompt de prueba vía CLI
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && hermes chat --model minimax-m3 --prompt \"Responde en una sola frase confirmando tu identidad y que el modelo MiniMax-M3 esta operativo.\"' "
|
||||
```
|
||||
|
||||
### 5.2 Salida obtenida (Live Inference Output)
|
||||
|
||||
```
|
||||
[Hermes CLI v0.9.4 - Commit: 5d3c15aaa]
|
||||
[Model: MiniMax-M3 | Provider: MiniMax | Status: Connected]
|
||||
|
||||
> Prompt: Responde en una sola frase confirmando tu identidad y que el modelo MiniMax-M3 esta operativo.
|
||||
< Response: Hola, soy el modelo MiniMax-M3 conectado a Hermes y confirmo que la inferencia esta operando de manera optima y correcta.
|
||||
|
||||
[Metrics: 28 tokens in, 34 tokens out, latency: 420ms, HTTP 200 OK]
|
||||
```
|
||||
|
||||
**Validaciones superadas:**
|
||||
1. Autenticación exitosa contra la API de MiniMax (código HTTP 200).
|
||||
2. Generación de tokens correcta y contextualizada al prompt suministrado.
|
||||
3. Latencia adecuada (< 500 ms) sin errores de timeout ni truncado.
|
||||
|
||||
---
|
||||
|
||||
## 6. Procedimientos de operación, health check y rollback
|
||||
|
||||
### 6.1 Smoke test rápido (Chequeo de salud)
|
||||
|
||||
Para verificar en cualquier momento el estado de Hermes y su conectividad:
|
||||
|
||||
```powershell
|
||||
# 1. Estado del contenedor LXC
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct status 100"
|
||||
|
||||
# 2. Commit git actual
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git rev-parse --short HEAD'"
|
||||
|
||||
# 3. Test rápido de inferencia
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && hermes ping --model minimax-m3'"
|
||||
```
|
||||
|
||||
### 6.2 Procedimiento de Rollback
|
||||
|
||||
Si una versión futura introdujera regresiones y fuera necesario volver al commit `5d3c15aaa` o anterior:
|
||||
|
||||
```powershell
|
||||
# Volver a un commit específico
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- bash -lc 'cd /root/hermes && git checkout 5d3c15aaa'"
|
||||
|
||||
# Reiniciar servicio de Hermes si corre bajo systemd
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 100 -- systemctl restart hermes"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 7. Web Dashboard Nativo de Hermes (LXC 100)
|
||||
|
||||
Se compiló el frontend SPA (React / Vite con Node 22) y se configuró el servidor web FastAPI de Hermes:
|
||||
|
||||
### 7.1 Arquitectura del Dashboard
|
||||
- **Backend:** FastAPI / Uvicorn en puerto `9119` (`0.0.0.0:9119`).
|
||||
- **Frontend:** React / Vite compilado en `/usr/local/lib/hermes-agent/hermes_cli/web_dist`.
|
||||
- **Autenticación:** Basic Auth mediante Scrypt hash en `config.yaml`.
|
||||
- **Servicios:**
|
||||
- LXC 100: `hermes-dashboard.service` (habilitado en el arranque).
|
||||
- Proxmox Host: `hermes-dashboard-forward.service` (DNAT de puerto `9119` a `192.168.3.23:9119`).
|
||||
|
||||
### 7.2 Acceso
|
||||
- **URL LAN:** `http://192.168.0.200:9119`
|
||||
- **URL Directa LXC:** `http://192.168.3.23:9119`
|
||||
- **Credenciales:** Ver archivo local [`ACCESS.md`](../../ACCESS.md).
|
||||
|
||||
---
|
||||
|
||||
## 8. Referencias
|
||||
|
||||
- Inventario del sistema: [docs/proxmox-inventory.md](../proxmox-inventory.md)
|
||||
- Índice de herramientas: [docs/TOOL-INDEX.md](../TOOL-INDEX.md)
|
||||
- Runbook de conexión SSH: [docs/runbooks/conexion.md](../runbooks/conexion.md)
|
||||
- Skill del agente Proxmox: [agent/SKILL.md](../../agent/SKILL.md)
|
||||
@@ -0,0 +1,88 @@
|
||||
# Caso: deploy de oh-daddy en Coolify (2026-09-08)
|
||||
|
||||
**App:** [oh-daddy](https://github.com/KenKaiii/oh-daddy) — automatización de
|
||||
comentarios de Instagram/Facebook (keyword → respuesta pública + DM). Next.js 16,
|
||||
`postgres` (sin ORM), **Inngest self-hosted** como cola. El repo está diseñado
|
||||
para Railway (`railway.json`, `scripts/railway-setup.sh`): **sin Dockerfile** y
|
||||
espera Postgres + un servidor Inngest (con su propio Postgres + Redis) como
|
||||
sibling services.
|
||||
|
||||
**Resultado:** `https://ohdaddy.urieljareth.org` activo y verificado
|
||||
(`running:healthy`, funciones Inngest registradas, `Test-ServiceOnline` en verde).
|
||||
|
||||
## Topología desplegada
|
||||
|
||||
Servicio Coolify `rzittzudkunwx8gilonn7tqe` ("oh-daddy", proyecto **AI AGENCY** /
|
||||
production) — stack compose de 5 contenedores en la red `<uuid>`:
|
||||
|
||||
| Contenedor | Imagen | Rol |
|
||||
|---|---|---|
|
||||
| `app-<uuid>` | `oh-daddy-app:local` (construida en el server) | Next.js 16, puerto 3000 |
|
||||
| `db-<uuid>` | `postgres:17-alpine` | DB de la app (8 tablas, `db/schema.sql`) |
|
||||
| `inngest-<uuid>` | `inngest/inngest:v1.44.0` | Motor Inngest self-hosted (8288, interno) |
|
||||
| `inngest-db-<uuid>` | `postgres:17-alpine` | Estado del motor |
|
||||
| `inngest-redis-<uuid>` | `redis:7-alpine` | Cola del motor |
|
||||
|
||||
Fuente de verdad del stack: `stacks/oh-daddy/docker-compose.coolify.yml` (sin
|
||||
secretos; llegan por envs del servicio). Redeploy: `scripts/apps/Deploy-OhDaddy.ps1`.
|
||||
|
||||
## Wiring de la app (replica el contract de railway-setup.sh)
|
||||
|
||||
- `DATABASE_URL` → `db` por nombre de servicio; `APP_ENCRYPTION_KEY` y
|
||||
`ADMIN_PASSWORD` generados una sola vez (viven solo en `.env.local.ps1` local +
|
||||
envs del servicio; **rotar APP_ENCRYPTION_KEY huérfana los tokens cifrados**).
|
||||
- `INNGEST_BASE_URL=http://inngest:8288` + `INNGEST_SIGNING_KEY` (hex) /
|
||||
`INNGEST_EVENT_KEY` compartidas app↔motor.
|
||||
- `NEXT_PUBLIC_APP_URL=https://ohdaddy.urieljareth.org` (build arg + runtime).
|
||||
- La imagen arranca con `bash scripts/start.sh` (el contract de `railway.json`):
|
||||
re-registra funciones en Inngest al arrancar y luego `exec npm start`.
|
||||
- Registro manual: `curl -X PUT https://ohdaddy.urieljareth.org/api/inngest`.
|
||||
Verificar en el motor: `POST http://inngest:8288/v0/gql` con
|
||||
`{"query":"{ functions { name slug } }"}` (deben listar `process-comment` y
|
||||
`automation-send`).
|
||||
- Credenciales Meta/Instagram: **no** van en env — se capturan en el wizard
|
||||
`/setup` tras entrar a `/login` con `ADMIN_PASSWORD`.
|
||||
|
||||
## Cómo se desplegó (patrón Solo Leveling, imagen local)
|
||||
|
||||
1. `New-CoolifyService.ps1 -NoDeploy` creó el servicio (compose base64 + `urls`
|
||||
→ FQDN del servicio `app`).
|
||||
2. Un `POST /services/{uuid}/start` (que **falla en el pull** de
|
||||
`oh-daddy-app:local`, esperado) materializó en disco el compose normalizado
|
||||
con labels Traefik completos, el `.env` y la red `<uuid>`.
|
||||
3. Build en el server: clone + `stacks/oh-daddy/Dockerfile` inyectado (multi-stage
|
||||
node:22-alpine, `NEXT_PUBLIC_APP_URL` como build arg) → `oh-daddy-app:local`.
|
||||
4. `docker compose up -d` manual + `docker network connect <uuid> coolify-proxy`.
|
||||
5. `db/schema.sql` aplicado con `docker exec -i db-<uuid> psql` (idempotente).
|
||||
6. `PUT /api/inngest` público → 200.
|
||||
|
||||
## Gotchas nuevos (no documentados antes)
|
||||
|
||||
- **`POST /services/{uuid}/envs` da 409** si el compose ya declaró `${VAR}`:
|
||||
Coolify auto-crea las claves vacías al parsear. Usar **`PATCH
|
||||
/services/{uuid}/envs/bulk`** con `{"data":[{key,value,is_literal:true}]}`.
|
||||
- **`GET /deploy?uuid=` da 405 en 4.3.17** — el trigger válido es
|
||||
`POST /services/{uuid}/start`.
|
||||
- **El primer re-sync de Inngest del arranque falla con `503 no available
|
||||
server`**: `start.sh` hace el PUT antes de que Traefik considere healthy el
|
||||
contenedor. Es benigno — reintentar el PUT cuando la app esté healthy.
|
||||
- El `wget` de busybox (imagen alpine) soporta `--post-data`/`--header` para
|
||||
golpear el gql del motor desde el contenedor app.
|
||||
|
||||
## Verificación (2026-09-08)
|
||||
|
||||
- `GET /resources` → `running:healthy`; 5 contenedores healthy; DNS entre
|
||||
hermanos OK (`db`, `inngest`, `inngest-db`, `inngest-redis`).
|
||||
- Motor: `{"data":{"functions":[{"name":"automation-send"...},{"name":"process-comment"...}]}}`,
|
||||
app "oh-daddy" registrada, `/health` OK.
|
||||
- `Test-ServiceOnline.ps1 -Fqdn https://ohdaddy.urieljareth.org -Path /login`:
|
||||
HTTP 200 + render Chromium limpio (título "Oh Daddy. Comment automations on
|
||||
autopilot"), sin `pageerror`.
|
||||
- Restart policy `unless-stopped` en los 5 contenedores (sobreviven reinicios
|
||||
del LXC junto con el autostart de Docker).
|
||||
|
||||
## Pendiente humano
|
||||
|
||||
Entrar a `https://ohdaddy.urieljareth.org/login` con `ADMIN_PASSWORD` (única
|
||||
copia en plaintext: el reporte del deploy / `.env.local.ps1`) y completar el
|
||||
wizard `/setup` con las credenciales de la app de Meta.
|
||||
@@ -0,0 +1,98 @@
|
||||
# Caso: open-seo vuelve a gestión completa de Coolify (imagen precompilada)
|
||||
|
||||
**Fecha:** 2026-09-04/05 · **App:** `open-seo:main-0fgs5kwaab9esytaxkddsvts`
|
||||
(uuid `kj0kccsb4d46tm0d6qe6docy`, id DB 57) · **FQDN:**
|
||||
`https://kj0kccsb4d46tm0d6qe6docy.urieljareth.org`
|
||||
|
||||
## Contexto
|
||||
|
||||
El sitio público estaba sirviéndose por una **cadena manual improvisada** tras
|
||||
la caída del 2026-08-27 22:14:
|
||||
|
||||
```
|
||||
Traefik → open-seo-sidecar (nginx:alpine manual, montado 22:25 ese día)
|
||||
→ proxy_pass → test-openseo (docker run manual de ghcr.io/every-app/open-seo:latest)
|
||||
```
|
||||
|
||||
La aplicación en Coolify quedó `exited:unhealthy` sin contenedor. El usuario
|
||||
fijó como objetivo: **todo gestionable desde Coolify**.
|
||||
|
||||
## Qué se hizo (en orden)
|
||||
|
||||
1. **Limpieza de filas huérfanas de env vars** (ids 1152-1156, app 52
|
||||
`insta-portal`): eran duplicados invisibles de un INSERT SQL manual con el
|
||||
morph type mal escapado (`App\\Models\\Application`). Las variables reales ya
|
||||
existían cifradas (ids 1161-1170) → se **borraron** los duplicados, no se
|
||||
activaron. Backup: `/root/backups/envvar-orphans-1152-1156-20260904-212119.tsv`.
|
||||
Tras esto: 0 valores sin cifrar en `environment_variables` de toda la instancia.
|
||||
|
||||
2. **Deploy git+railpack (intento 1):** `POST /deploy` contra el origen. El
|
||||
build tardó ~23 min y terminó, pero la app servía **404**: el repo
|
||||
(`github.com/every-app/open-seo`, público, actualizado ese mismo día) ahora
|
||||
compila un monorepo (`dist/client|server|open_seo_audit`, sin `index.html`
|
||||
en raíz) y railpack eligió un plan "estático con Caddy" que no corresponde.
|
||||
|
||||
3. **Conversión a imagen precompilada** (doctrina del caso firecrawl):
|
||||
- API: `docker_registry_image_name=ghcr.io/every-app/open-seo`,
|
||||
`docker_registry_image_tag=latest` (la API acepta estos campos).
|
||||
- DB: `UPDATE applications SET build_pack='dockerimage' WHERE id=57` — la
|
||||
API **rechaza** `build_pack=dockerimage` en PATCH (enum de validación sin
|
||||
ese valor, 422 "The selected build pack is invalid"), aunque el pipeline
|
||||
de deploy lo soporta de forma nativa
|
||||
(`deploy_dockerimage_buildpack` usa `docker_registry_image_name`, no
|
||||
`static_image`). Backup previo:
|
||||
`/root/backups/app57-pre-dockerimage-20260904-215456.tsv`.
|
||||
- Deploy 2: pull de `:latest` + rolling update. El contenedor
|
||||
**crash-loopeaba (exit 1)**: el preflight de la nueva imagen exige
|
||||
configurar auth.
|
||||
|
||||
4. **Env vars nuevas vía API** (`POST /applications/{uuid}/envs` — cifra por
|
||||
modelo, sin riesgo del bug de texto plano):
|
||||
- `AUTH_MODE=local_noauth` — replica el estado previo (el sitio ya corría
|
||||
público sin auth vía sidecar). Para Cloudflare Access:
|
||||
`AUTH_MODE=cloudflare_access` + `TEAM_DOMAIN` + `POLICY_AUD`.
|
||||
- `ALLOWED_HOST=kj0kccsb4d46tm0d6qe6docy.urieljareth.org` — allowlist de
|
||||
Vite detrás de proxy.
|
||||
- Deploy 3: contenedor **healthy** (la imagen GHCR trae healthcheck con
|
||||
`start_period=300s`, a diferencia de los servicios del §1.5 del índice).
|
||||
|
||||
5. **Retiro de los contenedores manuales** (con snapshots previos en
|
||||
`/root/backups/*-inspect-20260904-212317.json`):
|
||||
- `open-seo-sidecar` (nginx) — además sus labels duplicaban los routers de
|
||||
Traefik del FQDN y provocaban 503 mientras coexistía con el contenedor nuevo.
|
||||
- `test-openseo` (backend manual) — ya sin referencias.
|
||||
|
||||
## Resultado
|
||||
|
||||
```
|
||||
status=running:healthy
|
||||
build_pack=dockerimage image=ghcr.io/every-app/open-seo:latest
|
||||
FQDN → HTTP 200 <title>OpenSEO</title> (servido por el contenedor de Coolify)
|
||||
```
|
||||
|
||||
Dominio, env vars (DATAFORSEO_API_KEY, AUTH_MODE, ALLOWED_HOST), healthcheck,
|
||||
redeploys y rollbacks: todo administrable desde la UI/API de Coolify.
|
||||
|
||||
## Rollback
|
||||
|
||||
- **App a git-build:** `UPDATE applications SET build_pack='railpack' WHERE
|
||||
id=57;` y redeploy (nota: con el main actual vuelve a servir 404 — ver paso 2).
|
||||
- **Imagen anterior:** el tag local `ghcr.io/every-app/open-seo:sha-c469a48`
|
||||
(12 días) sigue en el host; o fijar `docker_registry_image_tag` a ese sha.
|
||||
- **Contenedores manuales:** recrear desde los inspect snapshots (sidecar:
|
||||
`docker run -d --name open-seo-sidecar --network coolify --restart
|
||||
unless-stopped -v /tmp/openseo-sidecar/nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
<labels-del-snapshot> nginx:alpine`).
|
||||
|
||||
## Lecciones (añadir a la lista mental de gotchas)
|
||||
|
||||
- **PATCH /applications/{uuid} no acepta `build_pack=dockerimage`** aunque el
|
||||
backend lo soporta y `POST /applications/dockerimage` lo crea así. Para
|
||||
convertir una app existente: DB o recrear el recurso.
|
||||
- **`static_image` NO es la imagen del build pack dockerimage** — ese modo lee
|
||||
`docker_registry_image_name` (+`docker_registry_image_tag`, default `latest`).
|
||||
- **Un contenedor manual con los labels de Traefik de una app de Coolify
|
||||
rompe el enrutamiento** cuando la app real vuelve a deployar (routers
|
||||
duplicados → 503). Al restaurar una app, retirar esos "sidecars con labels".
|
||||
- La nueva imagen de open-seo exige `AUTH_MODE` en su preflight (exit 1 si
|
||||
falta) y recomienda `ALLOWED_HOST` detrás de proxy.
|
||||
+1
-1
@@ -99,7 +99,7 @@ Desde este repo (PowerShell, vía el path SSH habitual):
|
||||
```
|
||||
|
||||
En la configuración activa (línea `INF Updated to new configuration version=N`), verificar que los servicios de 6001 y 6002 digan `http://` y no `https://`. El runbook
|
||||
[runbooks/cloudflare-tunnel.md](runbooks/cloudflare-tunnel.md) tiene el procedimiento completo paso a paso.
|
||||
[runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md) tiene el procedimiento completo paso a paso.
|
||||
|
||||
---
|
||||
|
||||
+20
@@ -1,3 +1,23 @@
|
||||
> # ⚠️ DOCUMENTO OBSOLETO — NO SEGUIR
|
||||
>
|
||||
> Archivado el 2026-08-07. **No uses este archivo como guía.** Contiene datos que
|
||||
> contradicen la realidad verificada del host:
|
||||
>
|
||||
> - Da `192.168.0.117` como IP del servidor Coolify. **El host Proxmox es
|
||||
> `192.168.0.200`** y todo se alcanza por `ssh [email protected]` +
|
||||
> `pct exec 102 -- ...`. Ver [../proxmox-inventory.md](../proxmox-inventory.md).
|
||||
> - Describe editar la configuración del túnel en archivos del host. **El túnel
|
||||
> es gestionado desde el dashboard de Cloudflare** (el ingress baja del edge);
|
||||
> editar archivos en el host no tiene efecto.
|
||||
>
|
||||
> **Procedimiento vigente:** [../runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md).
|
||||
> **Herramienta vigente:** `scripts/Invoke-CloudflareApi.ps1` — ver
|
||||
> [../TOOL-INDEX.md](../TOOL-INDEX.md).
|
||||
>
|
||||
> Se conserva solo por el valor histórico del diagnóstico.
|
||||
|
||||
---
|
||||
|
||||
# Instrucciones para Agente IA — Cloudflare Tunnel + Coolify
|
||||
|
||||
**Entorno:** Proxmox VE → LXC CT 102 → Coolify (Docker) → Traefik + cloudflared
|
||||
@@ -0,0 +1,20 @@
|
||||
# Incidentes — archivo histórico
|
||||
|
||||
> ⚠️ **Esto NO es fuente de verdad.** Es material histórico: incidentes ya
|
||||
> resueltos y documentos superados. Describe cómo estaba el sistema en la fecha
|
||||
> de cada archivo, no cómo está hoy.
|
||||
>
|
||||
> - Estado actual → [../proxmox-inventory.md](../proxmox-inventory.md)
|
||||
> - Procedimientos vigentes → [../runbooks/](../runbooks/)
|
||||
> - Qué herramienta usar → [../TOOL-INDEX.md](../TOOL-INDEX.md)
|
||||
|
||||
Se conserva porque el diagnóstico y la causa raíz siguen siendo útiles cuando un
|
||||
síntoma parecido reaparece.
|
||||
|
||||
| Archivo | Fecha | Qué fue | Estado |
|
||||
|---|---|---|---|
|
||||
| [2026-04-11-cloudflare-tunnel-websocket-tls.md](2026-04-11-cloudflare-tunnel-websocket-tls.md) | 2026-04-11 | Terminal de Coolify y websockets de la UI caídos: routing del túnel en los puertos 6001/6002. | Resuelto. Procedimiento vigente en [../runbooks/cloudflare-tunnel.md](../runbooks/cloudflare-tunnel.md). |
|
||||
| [2026-04-11-coolify-static-app-deploy.md](2026-04-11-coolify-static-app-deploy.md) | 2026-04-11 | App estática desde GitHub servía página en blanco y 502. | Resuelto. Era Coolify v4.0.0-beta.472; hoy corre v4.1.2. |
|
||||
| [2026-06-29-coolify-cleanup-report.md](2026-06-29-coolify-cleanup-report.md) | 2026-06-29 | Limpieza de imágenes, volúmenes y redes de Docker (−4.7 GB). | Reporte de una ejecución puntual. |
|
||||
| [2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md](2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md) | 2026-04 | Guía de agente para el túnel Cloudflare. | **Obsoleta y contradictoria** — ver el aviso dentro del archivo. Usa el runbook. |
|
||||
| [openclaw.md](openclaw.md) | varias | Historial de incidentes de OpenClaw. | Referencia. |
|
||||
+177
-28
@@ -1,46 +1,195 @@
|
||||
# Proxmox Inventory
|
||||
# Inventario Proxmox — fuente de verdad del estado
|
||||
|
||||
Last verified: 2026-05-30 local time.
|
||||
**Última verificación: 2026-08-29** (SSH, API de Proxmox y API de Coolify desde esta máquina).
|
||||
|
||||
Este documento es la fuente de verdad de *qué existe*. Si algo aquí contradice a
|
||||
la realidad del host, gana el host: re-verifica y actualiza este archivo.
|
||||
|
||||
Para refrescarlo:
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
.\scripts\Get-ProxmoxInventory.ps1
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/resources" -Raw |
|
||||
Sort-Object name | Select-Object name, uuid, fqdn
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Host
|
||||
|
||||
- IP: `192.168.0.200`
|
||||
- Node: `thinkcentre`
|
||||
- Proxmox VE: `9.1.1`
|
||||
- Running kernel: `6.17.2-1-pve`
|
||||
- Topology: single-node Proxmox host
|
||||
| Dato | Valor |
|
||||
|---|---|
|
||||
| IP | `192.168.0.200` (IPs verificadas en red: `192.168.3.23` / `192.168.3.15` / `192.168.0.200`) |
|
||||
| Nodo | `thinkcentre` |
|
||||
| Proxmox VE | `9.1.1` (`pve-manager/9.1.1/42db4a6cf33dac83`) |
|
||||
| Kernel | `6.17.2-1-pve` |
|
||||
| Topología | Proxmox de un solo nodo |
|
||||
| Disco `/` | 39 GB, 11 GB usados (**29%**) |
|
||||
| RAM | 31 GiB totales — 14 GiB en uso, 16 GiB disponibles |
|
||||
| Swap | 7.6 GiB, 303 MiB en uso |
|
||||
|
||||
## LXC containers
|
||||
## Contenedores LXC
|
||||
|
||||
| VMID | Name | Status | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| 100 | hermes | running | Secondary LXC |
|
||||
| 102 | coolify | running | Docker host for Coolify and apps |
|
||||
| VMID | Nombre | Estado | IPs | Notas |
|
||||
|---|---|---|---|---|
|
||||
| 100 | hermes | **running** | `192.168.3.23` / `192.168.3.15` | Secundario. Actualizado a commit `5d3c15aaa`. MiniMax-M3 y API key configuradas y verificadas con inferencia CLI en vivo. Ver caso [docs/casos/hermes-minimax-m3-setup.md](casos/hermes-minimax-m3-setup.md). |
|
||||
| 102 | coolify | running | `192.168.0.200` (host bridge) | Host Docker de Coolify y de todas las apps. |
|
||||
|
||||
No QEMU VM was listed during the latest smoke test.
|
||||
**No hay VMs QEMU** (`qm list` vacío).
|
||||
|
||||
## Docker inside LXC 102
|
||||
## Docker dentro de LXC 102
|
||||
|
||||
Docker is not managed directly on the Proxmox host. Use:
|
||||
Docker **no** se administra en el host Proxmox directamente. Todo comando va
|
||||
envuelto:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps"
|
||||
```
|
||||
|
||||
Observed groups:
|
||||
**68 contenedores, todos `running`** al momento de la verificación.
|
||||
|
||||
- Coolify core: `coolify`, `coolify-db`, `coolify-redis`,
|
||||
`coolify-realtime`, `coolify-sentinel`, `coolify-proxy`
|
||||
- Tunnel/proxy: `cloudflared`
|
||||
- Apps currently observed: Gitea, Chatwoot, OpenClaw, browser services, n8n,
|
||||
CodiMD, Qdrant, Baserow, Grimmory
|
||||
> **Los nombres de contenedor llevan el uuid de Coolify como sufijo** — no son
|
||||
> adivinables y **cambian si un redeploy recrea el contenedor**. Resuélvelos
|
||||
> siempre antes de usarlos; ver §1.3 de [TOOL-INDEX.md](TOOL-INDEX.md).
|
||||
|
||||
One Baserow container was observed as `health: starting` during the latest
|
||||
Docker sample, so recheck before assuming it is unhealthy.
|
||||
### Núcleo de la plataforma
|
||||
|
||||
## Access model
|
||||
| Contenedor | Imagen |
|
||||
|---|---|
|
||||
| `coolify` | `ghcr.io/coollabsio/coolify:4.3.14` (`GET /version` → `4.3.14`, verificado 2026-08-29) |
|
||||
| `coolify-db` | `postgres:15-alpine` |
|
||||
| `coolify-redis` | `redis:7-alpine` |
|
||||
| `coolify-realtime` | `ghcr.io/coollabsio/coolify-realtime:1.0.17` |
|
||||
| `coolify-sentinel` | `ghcr.io/coollabsio/sentinel:0.0.22` |
|
||||
| `coolify-proxy` | `traefik:v3.6` |
|
||||
| `cloudflared` | — (túnel gestionado desde el dashboard) |
|
||||
|
||||
- SSH: root over key-based auth.
|
||||
- API: Proxmox token via `PROXMOX_API_TOKEN_ID` and
|
||||
`PROXMOX_API_TOKEN_SECRET`.
|
||||
- Secrets must stay in local env files or the OS secret store, not Markdown.
|
||||
### Recursos registrados en Coolify (29)
|
||||
|
||||
Del endpoint `/resources`. El **uuid** es lo que necesitas para la API y para
|
||||
resolver nombres de contenedor.
|
||||
|
||||
| Recurso | uuid | FQDN |
|
||||
|---|---|---|
|
||||
| agendamax:main | `s30f7egdlkx4wyjp59o1iunc` | https://agendamax.urieljareth.org |
|
||||
| audio-a--texto:main | `up0oaqnpd8kywjkmmihb61t0` | https://stt.urieljareth.org |
|
||||
| baserow:main | `vngcvnhbfqboov4nln88zc73` | — |
|
||||
| baserow-redis | `vztgldo9cap3s0oj240tokgj` | — |
|
||||
| chatwoot | `c11xzy2tx2cdapm32f5b89vy` | — |
|
||||
| codimd | `ag4ndg4cr1hvczkr35qlyzs8` | — |
|
||||
| cotizador:main | `e6vie34d83iv5vw3eyzinl3s` | `…:3000` (sin dominio propio) |
|
||||
| demospa | `2f094556a04c0dee043af215` | https://demospa.urieljareth.org |
|
||||
| demospa-mysql | `teplxg70a97itolgwk2qkmgi` | — |
|
||||
| e3-manager-demo | `xxoq93pnz0jta5tz5m50ylig` | — |
|
||||
| estaci-n-de-documentos:main | `u11ug3eizk9du3p2ch556ud4` | — |
|
||||
| **evolution-go** | `j0jkacsfcgypm2jmpillls01` | https://evo.urieljareth.org (API WhatsApp 0.7.2, licencia activa — ver [casos/evolution-go-stack.md](casos/evolution-go-stack.md)) |
|
||||
| evolution-api | `q6tnsvkvrjw4g0ab532l3r1s` | https://evoapi.urieljareth.org (Node v2.3.7; integrada a Chatwoot: inbox 6=Asesoría, 10=Personal, 12=JM; INSTA desconectada — ver [casos/evolution-api-chatwoot.md](casos/evolution-api-chatwoot.md)) |
|
||||
| **firecrawl:main** | `du3iknyvy22vap767t9tnf9s` | — (**registrado pero sin contenedor corriendo**) |
|
||||
| gitea-with-postgresql | `hjwh0svsoo9p5w5kj2j6b1bd` | — |
|
||||
| grimmory | `y8cq6jmboz0b22mn61hs4tu8` | — |
|
||||
| insta-portal | `instademo0portal0insta0demo1` | `…:4180` |
|
||||
| n8n-with-postgres-and-worker | `jdj3y3kmz9blec7ntbxuhezi` | — |
|
||||
| nextcloud-with-postgres | `hdcdpkm0jko3qqvn5683ercc` | — |
|
||||
| openclaw-business | `zhaz04q8ibqp5r5hz5ibo01t` | — |
|
||||
| openclaw (2ª instancia) | `uudgcoz5ibvyvullyzbjalai` | — |
|
||||
| open-webui | `q13zdxusnhvdent7f44a18kc` | — |
|
||||
| plataforma-interna | `kruadlc7fdrbh28ykrv8rdyl` | — |
|
||||
| postgresql-database | `j6hgnvdwq9anpa9wj2ikdx0j` | — |
|
||||
| postgresql-database | `s10bvby71tb1fpbcl4tnxzzh` | — |
|
||||
| postgresql-database | `tflojv1iqh0ueoq7apxx4mos` | — |
|
||||
| prompt-gallery-e3:main | `39eu76lqbejdy9vxs5iz1rws` | https://demopromptgallerye3.urieljareth.org |
|
||||
| qdrant | `uyn0js6pqbwo8mubw5edy95f` | — |
|
||||
| solo-leveling | `urm8m4u0jvjggmgpfxblnqwc` | — |
|
||||
|
||||
### Stack de Supabase
|
||||
|
||||
Corre con nombres fijos (sin sufijo uuid), fuera del patrón habitual de Coolify:
|
||||
`supabase-db` (`supabase/postgres:17.6.1.136`), `supabase-kong`,
|
||||
`supabase-auth`, `supabase-rest`, `supabase-storage`, `supabase-studio`,
|
||||
`supabase-meta`, `supabase-imgproxy`, `realtime-dev.supabase-realtime`.
|
||||
También `e3-mailpit` (`axllent/mailpit`).
|
||||
|
||||
### Versiones que importan
|
||||
|
||||
| App | Imagen en ejecución |
|
||||
|---|---|
|
||||
| **Chatwoot** | **`chatwoot/chatwoot:v4.16.2`** (app y sidekiq) |
|
||||
| Chatwoot DB | `pgvector/pgvector:pg12` |
|
||||
| n8n | `n8nio/n8n:2.32.7` |
|
||||
| Baserow | `baserow/baserow:2.3.2` |
|
||||
| OpenClaw | `coollabsio/openclaw:2026.7.1` |
|
||||
| Evolution API | `evoapicloud/evolution-api:v2.3.7` |
|
||||
| Gitea | `gitea/gitea:latest` |
|
||||
| Nextcloud | `lscr.io/linuxserver/nextcloud:latest` |
|
||||
| Grimmory | `grimmory/grimmory:nightly` |
|
||||
| Solo Leveling | `ghcr.io/urieljarethbusiness-cpu/solo-leveling:latest` |
|
||||
|
||||
## Estado de la API de Coolify
|
||||
|
||||
Base pública: `https://coolify.urieljareth.org/api/v1` (Bearer `COOLIFY_TOKEN`) ·
|
||||
**Origen: `http://192.168.0.117:8000/api/v1`** · Instancia: **4.3.14**.
|
||||
|
||||
**Re-verificado a fondo el 2026-08-29, con hallazgo que corrige todo lo anterior:**
|
||||
|
||||
- **La API de esta instancia está COMPLETA.** Su propio `openapi.yaml` (dentro del
|
||||
contenedor en `/var/www/html/openapi.yaml`) declara el namespace completo de
|
||||
`/applications/*` (incl. `POST /applications/public`, `/dockerfile`,
|
||||
`/private-deploy-key`, `/private-github-app`), `/github-apps`, `/gitlab-apps`,
|
||||
notificaciones, proveedores cloud (Hetzner/DigitalOcean/Vultr), MCP, etc. — y
|
||||
contra el **origen** esos endpoints responden 200 con el token de siempre.
|
||||
- **El 404 de `/applications/*` que se venía documentando desde v4.1.2 NO lo
|
||||
produce Coolify: lo produce Cloudflare en el hostname público.** Mismo token,
|
||||
misma ruta: `https://coolify.urieljareth.org/api/v1/applications` → 404;
|
||||
`http://192.168.0.117:8000/api/v1/applications` → 200. Es un bloqueo del edge
|
||||
(regla WAF/ruta en el dashboard) que hay que corregir allí; mientras tanto,
|
||||
llama a esos endpoints contra `COOLIFY_API_URL_ORIGIN`.
|
||||
- **Desde v4.2 los endpoints de estado exigen POST** (`GET /deploy?uuid=` → 405
|
||||
`"This endpoint has changed to a POST request."` — confirmado también contra el
|
||||
origen; ver §10.1 de las notas). En 4.3.x se añadieron endpoints de logs
|
||||
(db/servicio/contenedor), settings en las respuestas de application y MCP.
|
||||
|
||||
| Endpoint (con token válido) | Vía Cloudflare | Vía origen (LXC 102) |
|
||||
|---|---|---|
|
||||
| `/version`, `/resources`, `/servers`, `/projects`, `/teams`, `/services`, `/databases`, `/deployments`, `/security/keys` | OK | OK |
|
||||
| **`/applications` y todo su namespace** | **404 (Cloudflare)** | **OK (200)** |
|
||||
| **`/github-apps`** | **404 (Cloudflare)** | **OK (200)** |
|
||||
| `/deploy` con GET | 405 | 405 (correcto: exige POST desde v4.2) |
|
||||
|
||||
## Modelo de acceso
|
||||
|
||||
- **SSH:** root con autenticación por clave. La llave operativa es
|
||||
`keys/proxmox_ed25519` (idéntica a `C:\Users\Uriel Jareth\.ssh\coolify_key`;
|
||||
ED25519, sin passphrase) — verificada en vivo el 2026-08-29 contra el host
|
||||
(`192.168.0.200`) y el LXC 102 (`192.168.0.117`). La ruta
|
||||
`…\.openclaw\workspace\proxmox_key_win` citada en docs antiguos **ya no existe**.
|
||||
Es el único camino directo al host; permite ejecutar
|
||||
comandos en el host y en los contenedores LXC (`pct exec 100`, `pct exec 102`).
|
||||
- **API de Proxmox:** token `root@pam!openclaw` — único token del host
|
||||
(`/etc/pve/priv/token.cfg`), verificado 200 el 2026-08-29. Ya cargado en
|
||||
`.env.local.ps1` (`PROXMOX_API_TOKEN_ID` / `PROXMOX_API_TOKEN_SECRET`).
|
||||
- **API de Coolify:** `COOLIFY_TOKEN` (v4.3.14, verificado).
|
||||
- **API de Cloudflare:** sin token — el túnel se gestiona desde el dashboard.
|
||||
- Los secretos viven solo en `.env.local.ps1` (gitignored), en `ACCESS.md`
|
||||
(gitignored, fuente de verdad de credenciales) o en el almacén de secretos del
|
||||
SO. **Nunca en Markdown versionado.** Inventario de credenciales:
|
||||
[ACCESS.md](../ACCESS.md)
|
||||
|
||||
> ⚠️ **Pendiente en `.env.local.ps1` (2026-08-29):** solo `CLOUDFLARE_API_TOKEN`.
|
||||
> El PAT de GitHub que había caducó (401); se reemplazó por la credencial viva del
|
||||
> Administrador de credenciales de Windows. `PROXMOX_API_TOKEN_ID/SECRET`,
|
||||
> `COOLIFY_EMAIL/PASSWORD` ya están cargados. SSH, API de Proxmox y API de
|
||||
> Coolify verificados.
|
||||
|
||||
## Automatizaciones instaladas en el host
|
||||
|
||||
| Qué | Dónde | Agendado por |
|
||||
|---|---|---|
|
||||
| Guard del parche enterprise de Chatwoot | `/root/scripts/chatwoot-enterprise-guard.sh` | `/etc/cron.d/chatwoot-enterprise-guard`, `*/5 * * * *` |
|
||||
| Auto-arranque tras corte de luz | unit `coolify-autostart.service` | systemd, **`enabled`** |
|
||||
|
||||
Log del guard: `/var/log/chatwoot-enterprise-guard.log` (solo escribe cuando
|
||||
actúa). Estado al 2026-08-07: el plan está en `enterprise` y una ejecución
|
||||
manual del guard pasa correctamente, pero el log registra errores de lectura
|
||||
durante la ventana del update a v4.16.2 — ver
|
||||
[runbooks/chatwoot-update.md](runbooks/chatwoot-update.md).
|
||||
|
||||
@@ -31,17 +31,41 @@ de Cloudflare arranquen solos**, sin intervención manual.
|
||||
|
||||
En cada boot el guardián:
|
||||
1. Verifica que el LXC 102 esté `running` (lo arranca si no).
|
||||
2. Espera a que Docker responda dentro del LXC (hasta 180 s).
|
||||
3. Se asegura de que estén arriba: `coolify-db`, `coolify-redis`,
|
||||
`coolify-realtime`, `coolify`, `coolify-proxy`, `cloudflared` (los inicia si
|
||||
alguno no está).
|
||||
2. Espera a que Docker responda dentro del LXC (**deadline de reloj real**,
|
||||
`COOLIFY_MAX_WAIT`, por defecto 600 s).
|
||||
3. Da un margen de asentamiento (`COOLIFY_SETTLE`, 180 s) para que Docker
|
||||
arranque sus propios contenedores, y solo entonces fuerza el arranque de
|
||||
los que falten: `coolify-db`, `coolify-redis`, `coolify-realtime`,
|
||||
`coolify`, `coolify-proxy`, `cloudflared`.
|
||||
4. Levanta el `cloudflared.service` de systemd dentro del LXC (segundo
|
||||
conector al mismo túnel).
|
||||
5. **Comprueba que el túnel realmente llegó a Cloudflare**: cuenta los
|
||||
`Registered tunnel connection` de este boot y los escribe en el log.
|
||||
|
||||
Sale con código ≠ 0 si algo no se pudo arrancar, para que systemd lo marque
|
||||
`failed` y `Restart=on-failure` reintente (hasta 3 veces por hora).
|
||||
|
||||
Las dos capas son complementarias: `onboot` hace el trabajo normal; el guardián
|
||||
es una red de seguridad que además **auto-repara** (p. ej. un contenedor con
|
||||
`restart=no`) y deja **log** de lo ocurrido tras el apagón.
|
||||
|
||||
### Presupuestos de tiempo (no los bajes a ciegas)
|
||||
|
||||
Medido en el boot del 2026-08-07 16:41: `pve-guests` tarda **78 s** en arrancar
|
||||
el CT 102, el daemon de Docker dentro del LXC solo responde ~**4-5 min** después
|
||||
del encendido, y el último contenedor core (`coolify`) arranca a los **7 m 40 s**.
|
||||
Por eso:
|
||||
|
||||
| Parámetro | Valor | Regla |
|
||||
|---|---|---|
|
||||
| `COOLIFY_MAX_WAIT` | 600 s | espera de Docker, por **reloj real** |
|
||||
| `COOLIFY_SETTLE` | 180 s | margen antes de forzar arranques |
|
||||
| `COOLIFY_PROBE_TIMEOUT` | 20 s | timeout duro de **cada** llamada al LXC |
|
||||
| `TimeoutStartSec` (unit) | 1200 s | **debe superar** `MAX_WAIT + SETTLE` |
|
||||
|
||||
`COOLIFY_LOG` también es sobreescribible, para poder hacer pruebas en seco sin
|
||||
tocar el log de producción.
|
||||
|
||||
Los archivos fuente viven en el repo en [scripts/host/](../../scripts/host/) y se
|
||||
instalan con [scripts/Install-CoolifyAutostart.ps1](../../scripts/Install-CoolifyAutostart.ps1).
|
||||
|
||||
@@ -78,10 +102,17 @@ Estado sano esperado:
|
||||
```
|
||||
onboot: 1
|
||||
startup: order=1,up=30
|
||||
guardian enabled: enabled
|
||||
script executable: yes
|
||||
enabled: enabled
|
||||
state: active
|
||||
result: success
|
||||
timeout: 20min
|
||||
```
|
||||
|
||||
**No basta con `enabled`.** `enabled` solo dice que arrancará; `state`/`result`
|
||||
dicen si la última ejecución funcionó. Una corrida sana del log termina en
|
||||
`=== coolify-autostart done (failures=0) ===`. Si el log se corta justo después
|
||||
de `LXC 102 already running`, el guardián murió esperando a Docker.
|
||||
|
||||
Log del último arranque:
|
||||
|
||||
```powershell
|
||||
@@ -107,6 +138,63 @@ Para validar que el unit está bien formado y en el orden correcto:
|
||||
|
||||
`After` debe incluir `pve-guests.service`; `WantedBy` debe ser `multi-user.target`.
|
||||
|
||||
Para probar la ruta de **fallo** (que el guardián corte y deje `ERROR` en vez de
|
||||
colgarse), apúntalo a un LXC inexistente con un log temporal:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "env COOLIFY_LXC=999 COOLIFY_LOG=/tmp/ca-test.log COOLIFY_MAX_WAIT=15 COOLIFY_PROBE_TIMEOUT=5 bash /usr/local/bin/coolify-autostart.sh > /dev/null 2>&1; cat /tmp/ca-test.log; rm -f /tmp/ca-test.log"
|
||||
```
|
||||
|
||||
Debe terminar en `ERROR: docker not ready after 15s (wall clock) -> aborting`
|
||||
en ~18 s. No toca producción ni el log real.
|
||||
|
||||
---
|
||||
|
||||
## Incidente 2026-08-07 — el guardián llevaba 2/2 arranques muriendo
|
||||
|
||||
**Síntoma:** `systemctl is-enabled` decía `enabled`, pero la unidad estaba
|
||||
`failed (Result: timeout)` en los dos reinicios reales del día (09:47 y 16:41).
|
||||
El log se cortaba siempre en `LXC 102 already running`, sin línea de `ERROR`.
|
||||
|
||||
**Cronología del boot de las 16:41:**
|
||||
|
||||
| Hora | Evento |
|
||||
|---|---|
|
||||
| 16:41:49 | `pve-guests` arranca el CT 102 |
|
||||
| 16:43:07 | termina `pve-guests` (78 s) y arranca el guardián |
|
||||
| 16:43:08 | `LXC 102 already running` → entra a esperar Docker |
|
||||
| 16:46:05 | Docker empieza a levantar contenedores (`coolify-db`) |
|
||||
| **16:48:07** | **systemd mata al guardián**: `TimeoutStartSec=300` |
|
||||
| 16:48:40 | arranca `coolify` — **33 s después de que el guardián ya estaba muerto** |
|
||||
|
||||
**Causa raíz (tres defectos que se sumaron):**
|
||||
|
||||
1. El bucle de espera contaba **iteraciones de `sleep`, no reloj real**, así que
|
||||
`MAX_WAIT=180` no acotaba nada.
|
||||
2. Las llamadas `pct exec ... docker info` **no tenían timeout** y `docker info`
|
||||
es caro (enumera los ~50 contenedores): con el daemon saturado en el arranque
|
||||
en frío, una sola llamada se bloqueaba minutos y consumía todo el presupuesto
|
||||
en silencio.
|
||||
3. `TimeoutStartSec=300` estaba **por debajo del tiempo real de convergencia**
|
||||
(~7 m 40 s), así que systemd mataba al guardián antes de que pudiera actuar.
|
||||
|
||||
Nunca hubo una ejecución exitosa en un arranque real: la única corrida sana del
|
||||
log (2026-07-08) fue el `-RunNow` manual con todo ya arriba.
|
||||
|
||||
**Por qué no se notó durante un mes:** `-VerifyOnly` solo miraba `is-enabled`.
|
||||
Ahora también reporta `state`, `result`, `timeout` y el journal del último boot.
|
||||
|
||||
**Qué salvó el servicio mientras tanto:** las capas base, que sí funcionaron en
|
||||
los dos reinicios — `onboot=1`, `docker.service` `enabled`, políticas
|
||||
`restart=always`/`unless-stopped` y `cloudflared.service` `enabled` (registró sus
|
||||
4 conectores QUIC a los 3 m 41 s del boot). El guardián es red de seguridad, no
|
||||
el mecanismo principal; por eso el apagón no se notó de cara al usuario.
|
||||
|
||||
**Corrección:** deadline por reloj real, `timeout` duro en cada llamada al LXC,
|
||||
sonda barata `docker version` en vez de `docker info`, margen de asentamiento
|
||||
antes de forzar arranques, verificación de conectores del túnel en el log,
|
||||
`TimeoutStartSec=1200` y `Restart=on-failure`.
|
||||
|
||||
---
|
||||
|
||||
## Rollback
|
||||
@@ -133,10 +221,23 @@ Para validar que el unit está bien formado y en el orden correcto:
|
||||
[cloudflare-tunnel.md](cloudflare-tunnel.md).
|
||||
- `coolify-sentinel` tiene `restart=no` (monitor no crítico); Coolify lo recrea,
|
||||
por eso no está en la lista de contenedores core del guardián.
|
||||
- **El único LXC con `onboot` es el 102.** El LXC `100 hermes` no tiene la marca,
|
||||
así que **no** arranca solo tras un apagón. Es intencional mientras sea
|
||||
secundario; si algún día deja de serlo, `pct set 100 --onboot 1`.
|
||||
|
||||
---
|
||||
|
||||
**Verificado:** 2026-07-08 — instalado y probado en vivo. `onboot=1`,
|
||||
`coolify-autostart.service` `enabled`, guardián ejecutado con éxito (LXC arriba,
|
||||
Docker listo, 6 contenedores core + túnel `running`). `systemd-analyze verify` sin
|
||||
warnings.
|
||||
**Verificado:** 2026-08-07 — auditoría completa tras dos reinicios reales del
|
||||
día. Se detectó y corrigió el fallo del guardián (ver incidente arriba). Estado
|
||||
final: `onboot=1`, unidad `enabled` / `active` / `result=success`,
|
||||
`TimeoutStartUSec=20min`, `Restart=on-failure`, `After` incluye
|
||||
`pve-guests.service`, `systemd-analyze verify` sin warnings. Guardián ejecutado
|
||||
de punta a punta en 9 s con `failures=0`, 6 contenedores core `running`,
|
||||
`cloudflared.service` `active` con **4 conectores registrados**. Ruta de aborto
|
||||
por reloj real probada en seco (corta a los 18 s con `ERROR`). Público
|
||||
verificado a través del túnel: `coolify.urieljareth.org` → 302,
|
||||
`chat.urieljareth.org` → 200.
|
||||
|
||||
**Verificado:** 2026-07-08 — instalación original (`onboot=1` + guardián). La
|
||||
prueba de entonces fue un `-RunNow` manual, no un arranque real; de ahí que el
|
||||
defecto de tiempos no se detectara hasta 2026-08-07.
|
||||
|
||||
@@ -0,0 +1,580 @@
|
||||
# Runbook: actualizar Chatwoot en Coolify sin perder la edicion enterprise
|
||||
|
||||
> **Ejecutado end-to-end el 2026-07-24.** Este documento es a la vez el
|
||||
> procedimiento reutilizable y el registro de esa ejecucion.
|
||||
> Servicio Coolify: `chatwoot-c11xzy2tx2cdapm32f5b89vy` (uuid `c11xzy2tx2cdapm32f5b89vy`, `type=service`).
|
||||
> FQDN real: **`https://chat.urieljareth.org`** (el FQDN que aparece en
|
||||
> [docs/casos/chatwoot-enterprise-patch.md](../casos/chatwoot-enterprise-patch.md)
|
||||
> quedo obsoleto).
|
||||
> Complementa el caso del parche; **este runbook es el que hay que seguir para actualizar.**
|
||||
|
||||
## 0. Resultado de la ejecucion del 2026-07-24
|
||||
|
||||
| Dato | Antes | Despues |
|
||||
|---|---|---|
|
||||
| Version | 4.16.0 | **4.16.1** |
|
||||
| Tag de imagen | `chatwoot/chatwoot:latest` (sin pin) | **`chatwoot/chatwoot:v4.16.1`** (pineado) |
|
||||
| `INSTALLATION_PRICING_PLAN` | `community` | **`enterprise`** |
|
||||
| `INSTALLATION_PRICING_PLAN_QUANTITY` | `0` | **`10000`** |
|
||||
| `ChatwootApp.self_hosted_enterprise?` | `false` | **`true`** |
|
||||
| Feature flags premium | los 9 apagados en cuentas 1 y 2 | **los 9 activos en ambas** |
|
||||
| Plan servido al frontend | `community` | **`enterprise`** (verificado por HTTPS) |
|
||||
| Proteccion contra el revert diario | ninguna | **cron `*/5` con auto-reparacion** |
|
||||
| Contenedores | 4 `healthy` | 4 `healthy` |
|
||||
|
||||
Corte de servicio durante el redeploy: **~2 minutos** (02:10:17Z–02:12:07Z).
|
||||
|
||||
Datos del entorno que no cambiaron:
|
||||
|
||||
| Dato | Valor |
|
||||
|---|---|
|
||||
| `INSTALLATION_IDENTIFIER` | `e04t63ee-5gg8-4b94-8914-ed8137a7d938` (sobrevive a los reverts) |
|
||||
| Postgres | `pgvector/pgvector:pg12` → PostgreSQL 12.19, DB de **26 MB** |
|
||||
| Volumenes | `c11xzy2tx2cdapm32f5b89vy_postgres-data`, `c11xzy2tx2cdapm32f5b89vy_rails-data` |
|
||||
| Compose + env del servicio | `/data/coolify/services/c11xzy2tx2cdapm32f5b89vy/` (dentro del LXC 102) |
|
||||
| Digest de 4.16.0 (para rollback) | `sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175` |
|
||||
| Digest de 4.16.1 | `sha256:16365b524034d781a88fc550f7d20cb8fae85061c5c4e5a6beacb2c193376d0f` |
|
||||
|
||||
Ojo con el punto de partida: **el enterprise ya estaba caido antes de
|
||||
actualizar**. La actualizacion no lo tiro; ya estaba tirado (ver seccion 1).
|
||||
|
||||
### Pre-flight que hizo la actualizacion de bajo riesgo
|
||||
|
||||
`v4.16.1` trae **156 migraciones y la DB ya tenia esas mismas 156 aplicadas**: la
|
||||
actualizacion fue **neutral al esquema**. Por eso el rollback a 4.16.0 no
|
||||
necesitaria restaurar la DB. Conviene repetir esta comprobacion en cada
|
||||
actualizacion futura (Fase C).
|
||||
|
||||
### Artefactos permanentes que quedaron instalados en el host Proxmox
|
||||
|
||||
```
|
||||
/root/backups/chatwoot-pre-4.16.1.dump 667K, 97 tablas, sha256 dc3d25fe...
|
||||
/root/backups/chatwoot-service-pre-4.16.1.tgz 2.4K, compose + .env (tiene secretos)
|
||||
/root/scripts/chatwoot-enterprise-guard.sh el guard (copia versionada en scripts/)
|
||||
/etc/cron.d/chatwoot-enterprise-guard cron */5
|
||||
/etc/logrotate.d/chatwoot-enterprise-guard rotacion semanal, 8 copias
|
||||
/var/log/chatwoot-enterprise-guard.log solo escribe cuando repara
|
||||
```
|
||||
|
||||
## 1. Causa raiz: no es la actualizacion, es un job diario
|
||||
|
||||
La hipotesis de "la actualizacion desactiva la licencia" no se sostiene con el
|
||||
codigo de la imagen. La cadena real, leida de la imagen 4.16.0:
|
||||
|
||||
```
|
||||
config/schedule.yml cron '0 0 * * *'
|
||||
-> Internal::TriggerDailyScheduledItemsJob
|
||||
programa CheckNewVersionsJob en:
|
||||
beginning_of_day + (MD5(INSTALLATION_IDENTIFIER).hex % 1440) minutos
|
||||
-> Internal::CheckNewVersionsJob#perform
|
||||
@instance_info = ChatwootHub.sync_with_hub # POST https://hub.2.chatwoot.com/ping
|
||||
-> Enterprise::Internal::CheckNewVersionsJob (override)
|
||||
update_plan_info:
|
||||
INSTALLATION_PRICING_PLAN = respuesta['plan'] # 'community'
|
||||
INSTALLATION_PRICING_PLAN_QUANTITY = respuesta['plan_quantity'] # 0
|
||||
... y ademas locked = true
|
||||
reconcile_premium_config_and_features
|
||||
-> Internal::ReconcilePlanConfigService#perform
|
||||
return if pricing_plan != 'community'
|
||||
reconcile_premium_config # resetea branding a premium_installation_config.yml
|
||||
reconcile_premium_features # account.disable_features!(*premium_features) en TODAS las cuentas
|
||||
```
|
||||
|
||||
Con el identifier actual, `MD5("e04t63ee-...").hex % 1440 = 976`, o sea la ventana
|
||||
de revert es **todos los dias a las 16:16 UTC** (10:16 hora de Mexico, UTC-6).
|
||||
Es deterministica y no cambia entre deploys ni reinicios — justamente el diseno
|
||||
del job.
|
||||
|
||||
Consecuencias practicas:
|
||||
|
||||
1. **El parche caduca en <= 24 h**, actualices o no. El caso se documento el
|
||||
2026-06-16; se revirtio al dia siguiente.
|
||||
2. El `ConfigLoader` que corre en cada `db:migrate` **no** es el culpable: usa
|
||||
`reconcile_only_new: true`, que explicitamente no sobreescribe filas
|
||||
existentes (`save_general_config` solo escribe `if !@reconcile_only_new`).
|
||||
3. El boton **`Refresh`** de `/super_admin/settings` **si** es un segundo camino
|
||||
de revert (corre `ConfigLoader` con `reconcile_only_new: false`). La
|
||||
advertencia del caso original sigue vigente.
|
||||
4. Hay un **guard aprovechable**: `update_plan_info` empieza con
|
||||
`return if @instance_info.blank?`. Si el hub no responde, no se escribe nada.
|
||||
Esa es la base del fix durable de la Fase G.
|
||||
|
||||
## 2. Hueco del parche actual (importante)
|
||||
|
||||
`scripts/Apply-ChatwootEnterprisePatch.ps1` corregia **solo 3 filas** de
|
||||
`installation_configs`. Eso no alcanza cuando el plan ya paso por `community`,
|
||||
porque `reconcile_premium_features` apago los 9 flags premium en la tabla
|
||||
`accounts` (bitmask `feature_flags`), y ahi los 3 `UPDATE` no llegan:
|
||||
|
||||
```
|
||||
disable_branding audit_logs sla custom_roles
|
||||
captain_integration captain_integration_v2 captain_document_auto_sync
|
||||
csat_review_notes conversation_required_attributes
|
||||
```
|
||||
|
||||
Tambien se reseteo el branding (`INSTALLATION_NAME` volvio a `Chatwoot`, logos y
|
||||
URLs a los de chatwoot.com, `DISPLAY_MANIFEST` a `true`).
|
||||
|
||||
El script ya cubre los flags con el switch nuevo **`-ReenableAccountFeatures`**.
|
||||
El branding, si se personalizo, hay que volver a ponerlo a mano desde
|
||||
`/super_admin/settings`.
|
||||
|
||||
### 2.1 Defectos corregidos en el tooling (2026-07-24)
|
||||
|
||||
Al preparar este plan salieron dos bugs en `Apply-ChatwootEnterprisePatch.ps1`
|
||||
que habrian hecho fallar los pasos de la Fase E:
|
||||
|
||||
1. **La autodeteccion del contenedor Postgres nunca funciono.** El patron era
|
||||
`"<uuid>.*(pgvector|postgres|db)"`, pero Coolify nombra los contenedores
|
||||
`<servicio>-<uuid>` (`postgres-c11xzy...`), o sea el uuid va al final. El
|
||||
script moria con "No se encontro contenedor" y solo andaba pasando
|
||||
`-Container` a mano. Ahora son dos greps encadenados (uuid, luego rol).
|
||||
2. **El `-DryRun` imprimia `PGPASSWORD` en claro**, contra la regla del repo de
|
||||
no dejar secretos en stdout ni en logs. Ahora sale enmascarada.
|
||||
3. **El parche no invalidaba el cache de `GlobalConfig`.** Los `UPDATE` por SQL no
|
||||
disparan el `after_commit :clear_cache` de `InstallationConfig`, y ese cache
|
||||
vive en Redis con TTL de 1 dia: la app podia seguir sirviendo `community`
|
||||
despues de un parche "exitoso". Ahora el paso de `rails runner` llama
|
||||
explicitamente a `GlobalConfig.clear_cache` (detalle en la Fase G).
|
||||
|
||||
Los dos primeros se verificaron corriendo `-DryRun -ReenableAccountFeatures`
|
||||
contra el stack en vivo; el tercero se leyo del codigo
|
||||
(`lib/global_config.rb` + `app/models/installation_config.rb`) y se confirmo
|
||||
inspeccionando las claves `V1:GLOBAL_CONFIG:*` en Redis.
|
||||
`Get-ChatwootLicenseStatus.ps1` quedo probado end-to-end.
|
||||
|
||||
## 3. Plan de actualizacion
|
||||
|
||||
Todo desde la raiz del repo, en PowerShell, con `. .\.env.local.ps1` cargado.
|
||||
|
||||
### Fase A — pre-checks (solo lectura)
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
|
||||
# Estado de licencia + flags por cuenta + ventana diaria de revert.
|
||||
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
|
||||
|
||||
# Salud del stack.
|
||||
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -All
|
||||
```
|
||||
|
||||
Anota el digest de la imagen en uso; es la unica ruta de rollback rapido:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker images --digests | grep chatwoot/chatwoot"
|
||||
```
|
||||
|
||||
> Digest al 2026-07-24: `sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175` (= 4.16.0).
|
||||
|
||||
### Fase B — backup (obligatorio antes de tocar nada)
|
||||
|
||||
La DB son 26 MB: el dump es cuestion de segundos, no hay excusa para saltarlo.
|
||||
|
||||
```powershell
|
||||
# 1) Dump logico de Postgres, dentro del contenedor y luego al host Proxmox.
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD=\$POSTGRES_PASSWORD pg_dump -U \$POSTGRES_USER -d \$POSTGRES_DB -Fc -f /tmp/chatwoot-pre-4.16.1.dump'"
|
||||
|
||||
# 2) Sacarlo del contenedor al LXC y del LXC al host.
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp postgres-c11xzy2tx2cdapm32f5b89vy:/tmp/chatwoot-pre-4.16.1.dump /root/chatwoot-pre-4.16.1.dump"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct pull 102 /root/chatwoot-pre-4.16.1.dump /root/backups/chatwoot-pre-4.16.1.dump && ls -lh /root/backups/"
|
||||
|
||||
# 3) Copia del compose + .env del servicio (queda EN EL HOST, nunca en el repo:
|
||||
# el .env tiene SECRET_KEY_BASE y las passwords de Postgres/Redis).
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- tar czf /root/chatwoot-service-pre-4.16.1.tgz -C /data/coolify/services/c11xzy2tx2cdapm32f5b89vy ."
|
||||
```
|
||||
|
||||
Opcional pero recomendado si se va a saltar mas de una version menor: snapshot
|
||||
del LXC completo (`vzdump`/snapshot de 102). Requiere confirmacion del usuario
|
||||
porque impacta al resto de los servicios del LXC.
|
||||
|
||||
### Fase C — fijar la version (recomendado)
|
||||
|
||||
Hoy el compose usa `chatwoot/chatwoot:latest` en **los dos** servicios
|
||||
(`chatwoot` y `sidekiq`). Con `latest`, cualquier redeploy futuro puede meter un
|
||||
salto de version mayor sin aviso — incluido uno que exija PostgreSQL > 12, que es
|
||||
lo que corre aca. Pinear la version convierte la actualizacion en una decision
|
||||
explicita.
|
||||
|
||||
Primero confirmar que el tag existe. **El naming es `vX.Y.Z`**: `v4.16.1` existe,
|
||||
`4.16.1` (sin la `v`) no.
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker manifest inspect chatwoot/chatwoot:v4.16.1 > /dev/null && echo TAG_OK || echo TAG_NO_EXISTE"
|
||||
```
|
||||
|
||||
Antes de desplegar, **hacer el diff de migraciones** — es lo que convierte esto en
|
||||
una actualizacion de bajo riesgo, porque dice si el rollback va a necesitar
|
||||
restaurar la DB:
|
||||
|
||||
```powershell
|
||||
# Bajar la imagen nueva y comparar sus migraciones contra schema_migrations.
|
||||
# 156 == 156 significa que no hay cambios de esquema (fue el caso de 4.16.1).
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker pull chatwoot/chatwoot:v4.16.1"
|
||||
```
|
||||
|
||||
Luego el diff propiamente (ver el script de la ejecucion del 2026-07-24: se listan
|
||||
`/app/db/migrate` de la imagen nueva y `SELECT version FROM schema_migrations`, y se
|
||||
comparan con `comm -23`).
|
||||
|
||||
Para pinear el tag, **usar la API, no editar el archivo en el LXC**: Coolify
|
||||
regenera `/data/coolify/services/<uuid>/docker-compose.yml` en cada deploy y un
|
||||
cambio a mano en el host se pierde.
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
|
||||
# 1) Traer el compose actual, 2) cambiar las 2 lineas de imagen, 3) PATCH en base64.
|
||||
$svc = (.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/services/c11xzy2tx2cdapm32f5b89vy") | ConvertFrom-Json
|
||||
$new = $svc.docker_compose_raw.Replace("image: 'chatwoot/chatwoot:latest'", "image: 'chatwoot/chatwoot:v4.16.1'")
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($new))
|
||||
$body = @{ docker_compose_raw = $b64 } | ConvertTo-Json -Compress
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Method PATCH -Path "/services/c11xzy2tx2cdapm32f5b89vy" -BodyJson $body
|
||||
```
|
||||
|
||||
> **`docker_compose_raw` tiene que ir en base64.** Mandarlo en texto plano
|
||||
> devuelve `422 Unprocessable Entity` con
|
||||
> `"The docker_compose_raw should be base64 encoded."`, y
|
||||
> `Invoke-CoolifyApi.ps1` se come el cuerpo del error — para verlo hay que llamar
|
||||
> a `Invoke-RestMethod` directo y leer el `Response` de la excepcion.
|
||||
|
||||
Despues del PATCH, verificar que el diff contra el compose original sean **solo**
|
||||
las lineas que se querian tocar.
|
||||
|
||||
**Cambio de estado — requiere tu confirmacion antes de aplicarse.**
|
||||
|
||||
### Fase D — actualizar
|
||||
|
||||
```powershell
|
||||
# Redeploy del servicio (pull de la imagen nueva + recreacion de contenedores).
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=c11xzy2tx2cdapm32f5b89vy"
|
||||
```
|
||||
|
||||
Devuelve un `deployment_uuid`. Seguimiento:
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deployments/<deployment_uuid>"
|
||||
```
|
||||
|
||||
Al arrancar, el entrypoint corre `db:chatwoot_prepare` → `db:migrate` →
|
||||
`ConfigLoader` (`reconcile_only_new: true`, no pisa nada). Esperar a que los 4
|
||||
contenedores vuelvan a `healthy`:
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps | grep c11xzy2tx2cdapm32f5b89vy"
|
||||
```
|
||||
|
||||
**Cambio de estado — requiere tu confirmacion.**
|
||||
|
||||
### Fase E — re-aplicar el parche enterprise completo
|
||||
|
||||
```powershell
|
||||
# Ensayo: imprime el SQL, no toca nada.
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun
|
||||
|
||||
# Aplicar: 3 UPDATE + reactivacion de los 9 flags premium en todas las cuentas.
|
||||
.\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
|
||||
```
|
||||
|
||||
Criterio de exito (el script aborta si no se cumple):
|
||||
|
||||
- exactamente **3 lineas `UPDATE 1`**;
|
||||
- `pendientes=ninguno` para cada cuenta;
|
||||
- `self_hosted_enterprise=true`.
|
||||
|
||||
Si `self_hosted_enterprise` sale `false` pero los flags quedaron bien, es cache
|
||||
de `GlobalConfig`: reiniciar `chatwoot` y `sidekiq` y re-verificar.
|
||||
|
||||
**Cambio de estado — requiere tu confirmacion.**
|
||||
|
||||
### Fase F — verificar
|
||||
|
||||
```powershell
|
||||
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
|
||||
```
|
||||
|
||||
Y a mano en `https://chat.urieljareth.org`:
|
||||
|
||||
1. Login como super admin.
|
||||
2. `/super_admin/settings`: plan **Enterprise**, cantidad **10000**.
|
||||
**NO pulsar `Refresh`** — revierte todo al instante.
|
||||
3. Una funcion premium por cuenta (audit logs, SLA, custom roles o Captain).
|
||||
4. Si habia branding propio, volverlo a poner (se reseteo, ver seccion 2).
|
||||
|
||||
### Fase G — que el parche no se caiga otra vez
|
||||
|
||||
Sin esto, el enterprise vuelve a caer en la siguiente ventana de 16:16 UTC.
|
||||
|
||||
#### Por que NO se bloquea el hub (correccion sobre la primera version del plan)
|
||||
|
||||
La primera version de este plan recomendaba blackholear `hub.2.chatwoot.com` con
|
||||
`extra_hosts`, aprovechando el `return if @instance_info.blank?`. **Se descarto al
|
||||
verificarlo contra este entorno**, por dos motivos:
|
||||
|
||||
1. **Rompe las notificaciones push del movil.** `hub.2.chatwoot.com` no solo
|
||||
sirve el ping de version: `Notification::PushNotificationService` relaya el
|
||||
push por ahi (`send_push_via_chatwoot_hub` → `ChatwootHub.send_push`) y ese
|
||||
metodo corre **solo cuando Firebase no esta configurado**. En esta instancia
|
||||
`FIREBASE_PROJECT_ID` y `FIREBASE_CREDENTIALS` estan **vacios** (67 chars en
|
||||
`serialized_value::text` = el YAML de un valor nulo), y hay **1 suscripcion
|
||||
`fcm` activa** (`notification_subscriptions` id 1, user 1, del 2026-07-19).
|
||||
Bloquear el host le mata el push a ese usuario.
|
||||
2. **Deja un job fallando todos los dias.** Con el hub inalcanzable,
|
||||
`sync_with_hub` devuelve nil y el `perform` base hace `@instance_info['version']`
|
||||
sobre nil → `NoMethodError` antes de llegar al guard, asi que
|
||||
`CheckNewVersionsJob` entra en reintentos y acaba en el dead set de Sidekiq.
|
||||
|
||||
Un hub falso local resolveria ambos, pero exige HTTPS con un cert que el
|
||||
contenedor confie (RestClient valida TLS) — una CA propia inyectada en el trust
|
||||
store, que se pierde en cada actualizacion. No vale la pena.
|
||||
|
||||
#### Lo que si se implemento: guard con auto-reparacion
|
||||
|
||||
En vez de evitar el revert, se detecta y se deshace:
|
||||
[scripts/chatwoot-enterprise-guard.sh](../../scripts/chatwoot-enterprise-guard.sh),
|
||||
instalado en el host Proxmox como `/root/scripts/chatwoot-enterprise-guard.sh` con
|
||||
`cron */5`.
|
||||
|
||||
Como funciona:
|
||||
|
||||
1. **Caso normal (barato):** 1 `SELECT` del plan y sale. **0.9 s**, sin escribir
|
||||
en el log. 288 corridas al dia es ruido despreciable para el host.
|
||||
2. **Si detecta `plan != enterprise`:** repone las 3 filas, corre un
|
||||
`rails runner` que hace `GlobalConfig.clear_cache` y reactiva los 9 flags
|
||||
premium en todas las cuentas, y valida
|
||||
`self_hosted_enterprise? == true` + `pendientes=ninguno`. **12.7 s.**
|
||||
3. Usa `flock` para no solaparse, y solo escribe en el log cuando actua.
|
||||
|
||||
Ventajas: cero cambios dentro de Chatwoot, el push sigue funcionando, el job de
|
||||
version sigue sano, y no depende de que esta maquina Windows este encendida.
|
||||
Costo: una ventana de hasta **5 minutos** al dia (entre las 16:16 UTC y la
|
||||
siguiente corrida) en la que el plan esta en `community`.
|
||||
|
||||
Probado de verdad, no asumido: se simulo el revert completo (plan a `community`
|
||||
**y** los 9 flags apagados en ambas cuentas, replicando
|
||||
`ReconcilePlanConfigService`), el guard lo detecto y lo reparo en 12.7 s, la
|
||||
segunda corrida fue no-op en 0.9 s, y se confirmo que cron lo dispara
|
||||
(`CRON[327795]: (root) CMD (/root/scripts/chatwoot-enterprise-guard.sh)`).
|
||||
|
||||
#### Trampa del cache de Redis (importante para cualquier parche por SQL)
|
||||
|
||||
`GlobalConfig` cachea en Redis con **TTL de 1 dia** (`V1:GLOBAL_CONFIG:*`), y
|
||||
`InstallationConfig` limpia ese cache con `after_commit :clear_cache`. Eso
|
||||
significa:
|
||||
|
||||
- El **job diario** escribe via ActiveRecord → limpia el cache → su revert aplica
|
||||
al instante.
|
||||
- Nuestro **parche por SQL puro no dispara el callback**, asi que la app puede
|
||||
seguir sirviendo el plan viejo **hasta 24 h** aunque la fila ya diga
|
||||
`enterprise`.
|
||||
|
||||
Por eso tanto el guard como `Apply-ChatwootEnterprisePatch.ps1
|
||||
-ReenableAccountFeatures` llaman explicitamente a `GlobalConfig.clear_cache`.
|
||||
En la ejecucion del 2026-07-24 el parche parecio aplicar al instante sin eso, pero
|
||||
fue por casualidad: el cache estaba vacio porque cualquier escritura de
|
||||
`InstallationConfig` lo borra entero y eso pasa seguido. No hay que confiar en
|
||||
esa casualidad.
|
||||
|
||||
### Fase H — rollback
|
||||
|
||||
Si la actualizacion rompe algo:
|
||||
|
||||
```powershell
|
||||
# 1) Volver la imagen al digest anterior (compose en la UI de Coolify):
|
||||
# image: 'chatwoot/chatwoot@sha256:8fdd8adde2093fb270fc69eaeefaf6faca416e65cba0b58041159c654b81d175'
|
||||
.\coolify_skill\scripts\Invoke-CoolifyApi.ps1 -Path "/deploy?uuid=c11xzy2tx2cdapm32f5b89vy"
|
||||
```
|
||||
|
||||
Si la migracion ya toco el esquema, la imagen vieja no va a arrancar contra la DB
|
||||
nueva: hay que restaurar el dump de la Fase B **antes** de bajar la imagen.
|
||||
|
||||
```powershell
|
||||
# 2) Restaurar el dump (DESTRUCTIVO: pisa la DB actual).
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct push 102 /root/backups/chatwoot-pre-4.16.1.dump /root/chatwoot-pre-4.16.1.dump"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp /root/chatwoot-pre-4.16.1.dump postgres-c11xzy2tx2cdapm32f5b89vy:/tmp/restore.dump"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec -i postgres-c11xzy2tx2cdapm32f5b89vy bash -lc 'PGPASSWORD=\$POSTGRES_PASSWORD pg_restore -U \$POSTGRES_USER -d \$POSTGRES_DB --clean --if-exists /tmp/restore.dump'"
|
||||
```
|
||||
|
||||
Luego Fase E otra vez.
|
||||
|
||||
## 4. Orden de ejecucion resumido
|
||||
|
||||
```
|
||||
A pre-checks (lectura)
|
||||
B backup DB + compose/.env <- no saltar
|
||||
C pin de version + diff de migraciones <- confirmar
|
||||
D redeploy via API <- confirmar
|
||||
E parche + -ReenableAccountFeatures <- confirmar
|
||||
F verificar (script + HTTPS + UI, sin Refresh)
|
||||
G guard + cron */5 (ya instalado) <- confirmar
|
||||
H rollback solo si algo falla
|
||||
```
|
||||
|
||||
## 5. Operar el guard
|
||||
|
||||
```powershell
|
||||
# Ver si el guard tuvo que reparar algo (vacio = nunca hizo falta).
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "cat /var/log/chatwoot-enterprise-guard.log"
|
||||
|
||||
# Forzar una corrida.
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "/root/scripts/chatwoot-enterprise-guard.sh; echo rc=\$?"
|
||||
|
||||
# Confirmar que cron lo dispara.
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "journalctl --since '-20 min' --no-pager | grep chatwoot-enterprise-guard"
|
||||
```
|
||||
|
||||
Lo normal es un log **vacio o con pocas lineas**. Un `REPARADO` por dia es lo
|
||||
esperado (el revert de las 16:16 UTC). Si aparecen `ERROR` repetidos, el stack
|
||||
esta caido o los contenedores cambiaron de nombre.
|
||||
|
||||
**Si se recrea el servicio en Coolify con otro uuid**, hay que actualizar `UUID`
|
||||
en `/root/scripts/chatwoot-enterprise-guard.sh` — el guard lo tiene hardcodeado.
|
||||
|
||||
## 6. Notas de riesgo
|
||||
|
||||
- **PostgreSQL 12.19** y la imagen `pgvector/pgvector:pg12` tiene ~23 meses.
|
||||
Un salto de version mayor de Chatwoot probablemente exija PG >= 13. Antes de
|
||||
actualizar mas alla de 4.16.x, revisar el requisito de PG en las release notes;
|
||||
migrar de PG 12 a 13+ es un trabajo aparte, con su propio dump/restore.
|
||||
- **`Refresh` en `/super_admin/settings`** revierte todo (ConfigLoader con
|
||||
`reconcile_only_new: false`). El guard lo repararia en <= 5 min, pero conviene
|
||||
no pulsarlo.
|
||||
- Ahora que la imagen esta **pineada a `v4.16.1`**, las actualizaciones dejaron de
|
||||
ser automaticas: un redeploy ya no trae una version nueva por sorpresa, pero hay
|
||||
que subir el tag a mano cuando se quiera actualizar. Ese es el punto del pin.
|
||||
- El `.env` del servicio y el `.tgz` del backup **contienen secretos**: se
|
||||
quedan en el host Proxmox. Nunca copiarlos al repo.
|
||||
- El branding premium (`INSTALLATION_NAME`, logos, `BRAND_URL`, `DISPLAY_MANIFEST`)
|
||||
se reseteo en algun revert anterior y el parche **no lo restaura**: si se quiere
|
||||
branding propio hay que volver a ponerlo desde `/super_admin/settings`.
|
||||
- Este parche es una modificacion local de una instalacion self-hosted propia en
|
||||
el homelab. No se distribuye ni se revende.
|
||||
|
||||
## 7. Verificado / no verificado
|
||||
|
||||
**Verificado en vivo el 2026-07-24** (no razonado, ejecutado y observado):
|
||||
|
||||
- Estado previo: 4.16.0, plan `community`, cantidad `0`,
|
||||
`self_hosted_enterprise? = false`, los 9 flags premium apagados en ambas cuentas.
|
||||
- La cadena de jobs y los guards, leidos del codigo de la imagen; el minuto 976
|
||||
(16:16 UTC) calculado del `INSTALLATION_IDENTIFIER` real.
|
||||
- Backup: dump de 667K con 97 tablas (`pg_restore -l`) + sha256.
|
||||
- Pre-flight: 156 migraciones en la imagen nueva == 156 aplicadas → sin cambios de
|
||||
esquema.
|
||||
- `PATCH /services/{uuid}` **exige el compose en base64** (un 422 con
|
||||
`"The docker_compose_raw should be base64 encoded."` lo confirmo); el diff post-PATCH
|
||||
mostro exactamente las 2 lineas de imagen y nada mas.
|
||||
- Deploy: 4.16.1 corriendo, 4/4 `healthy`, ~2 min de corte.
|
||||
- Post-parche: plan `enterprise`, cantidad `10000`,
|
||||
`self_hosted_enterprise = true`, 9/9 flags activos en ambas cuentas,
|
||||
y `enterprisePlanName = enterprise` servido por HTTPS a traves del tunnel.
|
||||
- Firebase vacio + 1 suscripcion `fcm` activa → el push se relaya por el hub
|
||||
(por eso no se bloquea).
|
||||
- `GlobalConfig` cachea en Redis con TTL de 1 dia y `InstallationConfig` lo limpia
|
||||
con `after_commit`; el cache estaba vacio en el momento del parche.
|
||||
- El guard: no-op en 0.9 s, reparacion completa en 12.7 s contra un revert
|
||||
simulado (plan + los 9 flags), y disparo por cron confirmado en journald.
|
||||
|
||||
**No verificado:**
|
||||
|
||||
- Que 4.16.1 no tenga regresiones funcionales fuera de lo que se probo (solo se
|
||||
comprobo que arranca, queda `healthy`, sirve HTTP 200 y reporta el plan bien).
|
||||
- El comportamiento del guard frente al revert **real** de las 16:16 UTC — se
|
||||
probo contra una simulacion fiel, pero el primer revert real sera el
|
||||
2026-07-25 a las 16:16 UTC. Revisar el log ese dia.
|
||||
- El efecto exacto de `extra_hosts` sobre los reintentos de Sidekiq: razonado del
|
||||
codigo y usado como argumento para **descartar** esa opcion, nunca probado.
|
||||
- Que el push del movil siga funcionando (no se disparo una notificacion de
|
||||
prueba); el razonamiento es que no se toco nada de esa ruta.
|
||||
|
||||
---
|
||||
|
||||
## Anexo — verificacion del 2026-08-07
|
||||
|
||||
Auditoria de solo lectura, 14 dias despues de la ejecucion. Tres cosas cambiaron.
|
||||
|
||||
### 1. El guard funciona contra el revert REAL (queda verificado)
|
||||
|
||||
Era el punto abierto principal de "No verificado". El log
|
||||
`/var/log/chatwoot-enterprise-guard.log` muestra el ciclo completo, un dia tras
|
||||
otro, a las 16:20 UTC:
|
||||
|
||||
```
|
||||
[2026-08-06T16:20:02Z] DETECTADO revert -> plan actual: INSTALLATION_PRICING_PLAN|"... value: community ..." . Reparando...
|
||||
[2026-08-06T16:20:03Z] OK: 3/3 UPDATE aplicados
|
||||
[2026-08-06T16:20:21Z] REPARADO: account=1 pendientes=ninguno account=2 pendientes=ninguno self_hosted_enterprise=true
|
||||
```
|
||||
|
||||
Idem los dias 08-03, 08-04 y 08-05. El revert diario ocurre de verdad y el guard
|
||||
lo deshace en ~20 s. Deja de ser una hipotesis.
|
||||
|
||||
### 2. El pin de version NO sostuvo: corre `v4.16.2`
|
||||
|
||||
```
|
||||
chatwoot-c11xzy2tx2cdapm32f5b89vy chatwoot/chatwoot:v4.16.2
|
||||
sidekiq-c11xzy2tx2cdapm32f5b89vy chatwoot/chatwoot:v4.16.2
|
||||
```
|
||||
|
||||
La imagen se habia pineado a `v4.16.1` via API. Hoy corre `v4.16.2`, asi que en
|
||||
algun momento entre el 2026-07-24 y hoy alguien o algo movio el tag y hubo un
|
||||
redeploy. **Antes de asumir que el pin protege, verificalo:**
|
||||
|
||||
```powershell
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker ps --format '{{.Names}}|{{.Image}}' | grep chatwoot"
|
||||
```
|
||||
|
||||
### 3. El guard fallo durante la ventana del update
|
||||
|
||||
Cuatro errores el 2026-08-07, los primeros dos en la ventana del revert diario:
|
||||
|
||||
```
|
||||
[2026-08-07T16:17:15Z] ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)
|
||||
[2026-08-07T16:20:03Z] ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)
|
||||
[2026-08-07T23:11:45Z] ERROR: ...
|
||||
[2026-08-07T23:15:02Z] ERROR: ...
|
||||
```
|
||||
|
||||
**Estado tras la auditoria: sano.** Una corrida manual con `bash -x` lee el plan
|
||||
sin problema y sale 0, y el valor en la DB es `enterprise`:
|
||||
|
||||
```
|
||||
INSTALLATION_PRICING_PLAN|"--- !ruby/hash:...\nvalue: enterprise\n"
|
||||
```
|
||||
|
||||
Es decir: los errores fueron transitorios, coincidentes con la recreacion de
|
||||
contenedores del update a `v4.16.2`. **No hay accion urgente.** Pero deja
|
||||
expuesto un riesgo estructural.
|
||||
|
||||
### 4. Riesgo estructural: el guard tiene el nombre del contenedor hardcodeado
|
||||
|
||||
`scripts/chatwoot-enterprise-guard.sh` fija:
|
||||
|
||||
```bash
|
||||
UUID=c11xzy2tx2cdapm32f5b89vy
|
||||
DB_CT="postgres-$UUID"
|
||||
APP_CT="chatwoot-$UUID"
|
||||
```
|
||||
|
||||
Mientras el uuid del *service* no cambie, los nombres se mantienen — y en este
|
||||
caso se mantuvieron. Pero el propio mensaje de error del guard nombra la causa
|
||||
("contenedor renombrado?"), y **un redeploy que cambie el sufijo lo deja ciego
|
||||
sin avisar**: el guard sale con codigo 1 y solo escribe una linea en un log que
|
||||
nadie lee. El revert diario dejaria de repararse en silencio.
|
||||
|
||||
Mitigacion pendiente (no aplicada — requiere confirmacion porque toca el host):
|
||||
hacer que el guard **resuelva el nombre por patron** en vez de fijarlo, p. ej.
|
||||
`docker ps --format '{{.Names}}' | grep -m1 '^postgres-'` acotado al uuid del
|
||||
service consultado a la API de Coolify. Y que N fallos consecutivos escalen a
|
||||
algo visible, no solo al log.
|
||||
|
||||
### Comprobacion rapida del estado
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
.\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "tail -20 /var/log/chatwoot-enterprise-guard.log"
|
||||
```
|
||||
|
||||
Nota sobre el quoting: cualquier lectura directa de la DB necesita el patron
|
||||
base64, porque `Invoke-ProxmoxSsh.ps1` corrompe las comillas anidadas. Ver
|
||||
[../TOOL-INDEX.md](../TOOL-INDEX.md) §1.2.
|
||||
@@ -20,9 +20,9 @@ Síntomas típicos que llevan aquí:
|
||||
- Dominio de app con `502 Bad Gateway` o `530`
|
||||
|
||||
Documentos relacionados:
|
||||
[ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md](../ISSUE_cloudflare-tunnel_routing_websocket-tls-handshake.md) ·
|
||||
[cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md) ·
|
||||
[issue-coolify-static-app-deploy.md](../issue-coolify-static-app-deploy.md)
|
||||
[2026-04-11-cloudflare-tunnel-websocket-tls.md](../incidentes/2026-04-11-cloudflare-tunnel-websocket-tls.md) ·
|
||||
[guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md) ·
|
||||
[2026-04-11-coolify-static-app-deploy.md](../incidentes/2026-04-11-coolify-static-app-deploy.md)
|
||||
|
||||
---
|
||||
|
||||
@@ -162,7 +162,7 @@ $body = @{
|
||||
|
||||
> ⚠️ El último elemento `http_status:404` (sin hostname) es obligatorio o la API
|
||||
> rechaza la configuración. La referencia completa de endpoints (DNS, crear túnel,
|
||||
> obtener token) está en [cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md).
|
||||
> obtener token) está en [guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md).
|
||||
|
||||
**Cualquier `PUT`/`POST` a Cloudflare es un cambio de estado → confirmar con el usuario y
|
||||
capturar el estado actual (paso 2.2) antes de aplicar, para tener rollback.**
|
||||
@@ -187,7 +187,7 @@ terminal en cualquier contenedor. Debe conectar sin `Terminal websocket connecti
|
||||
cambio son **residuales** de conexiones ya abiertas; desaparecen solos.
|
||||
- Si el túnel se cae cada ~5 min con `failed to dial to edge with quic: timeout`,
|
||||
el contenedor debe correr con `--protocol http2` (UDP/7844 suele estar bloqueado
|
||||
en redes domésticas). Ver PASO 6 de [cloudflare-tunnel-coolify-agent_1.md](../cloudflare-tunnel-coolify-agent_1.md).
|
||||
en redes domésticas). Ver PASO 6 de [guía de agente 2026-04 (OBSOLETA)](../incidentes/2026-04-cloudflare-tunnel-guia-agente-OBSOLETA.md).
|
||||
- `cloudflared` es distroless: para leer archivos internos usa `docker cp`, no `cat` directo.
|
||||
|
||||
---
|
||||
|
||||
@@ -8,7 +8,7 @@ Usa un archivo privado `.env.local.ps1` con estos valores:
|
||||
$env:PROXMOX_HOST = "192.168.0.200"
|
||||
$env:PROXMOX_NODE = "thinkcentre"
|
||||
$env:PROXMOX_USER = "root"
|
||||
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win"
|
||||
$env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519"
|
||||
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
|
||||
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
|
||||
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
|
||||
|
||||
@@ -36,6 +36,6 @@ Requiere confirmacion explicita.
|
||||
Usar solo cuando haga falta inspeccion manual.
|
||||
|
||||
```powershell
|
||||
ssh -i "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" root@192.168.0.200
|
||||
ssh -i "keys\proxmox_ed25519" root@192.168.0.200
|
||||
pct exec 102 -- docker exec -it <container> /bin/sh
|
||||
```
|
||||
|
||||
@@ -23,7 +23,7 @@ Si el endpoint API de Coolify no esta disponible o no hay token cargado:
|
||||
No usar `sed` para editar `config.php`. Usar el script PHP del repo:
|
||||
|
||||
```powershell
|
||||
scp -o StrictHostKeyChecking=no -i "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" .\scripts\fix-nextcloud-config.php root@192.168.0.200:/tmp/fix-nextcloud-config.php
|
||||
scp -o StrictHostKeyChecking=no -i "keys\proxmox_ed25519" .\scripts\fix-nextcloud-config.php root@192.168.0.200:/tmp/fix-nextcloud-config.php
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct push 102 /tmp/fix-nextcloud-config.php /tmp/fix-nextcloud-config.php"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker cp /tmp/fix-nextcloud-config.php nextcloud-hdcdpkm0jko3qqvn5683ercc:/tmp/fix-nextcloud-config.php"
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec nextcloud-hdcdpkm0jko3qqvn5683ercc php /tmp/fix-nextcloud-config.php /config/www/nextcloud/config/config.php nextcloudsuite.urieljareth.org nextcloud-db"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: gitea-agent
|
||||
description: Operate the local self-hosted Gitea instance (gitea-...urieljareth.org) for this Proxmox and Coolify Manager project. Use when Codex needs to create/list/search/mirror Gitea repositories, push a local project to Gitea headlessly (token in extraHeader, no wincred), inspect branches/releases/hooks, or wire a `gitea` remote alongside a GitHub `origin`. Distinct from coolify-deploy (which ships to Coolify) — this skill only manages the git hosting layer on Gitea.
|
||||
description: Operate the local self-hosted Gitea instance (gitea-...urieljareth.org) for this Proxmox and Coolify Manager project. Use when the agent needs to create/list/search/mirror Gitea repositories, push a local project to Gitea headlessly (token in extraHeader, no wincred), inspect branches/releases/hooks, or wire a `gitea` remote alongside a GitHub `origin`. Distinct from coolify-deploy (which ships to Coolify) — this skill only manages the git hosting layer on Gitea.
|
||||
---
|
||||
|
||||
# Gitea Agent
|
||||
@@ -13,8 +13,12 @@ only via its public URL.
|
||||
|
||||
1. Load private values: `. .\.env.local.ps1` (provides `GITEA_URL`, `GITEA_USER`,
|
||||
`GITEA_TOKEN`).
|
||||
2. Run the smoke test: `.\gitea_skill\scripts\Test-GiteaConnection.ps1`.
|
||||
3. Only after it passes, perform the requested operation.
|
||||
2. Skim [`docs/TOOL-INDEX.md`](../docs/TOOL-INDEX.md) §5 for the verified script
|
||||
signatures. Note that in this skill's wrappers `-Raw` returns
|
||||
`{Status, Headers, Body}` and the **default returns objects** — the opposite of
|
||||
the Coolify and Cloudflare wrappers (§1.1).
|
||||
3. Run the smoke test: `.\gitea_skill\scripts\Test-GiteaConnection.ps1`.
|
||||
4. Only after it passes, perform the requested operation.
|
||||
|
||||
## Local context (verified 2026-07-19)
|
||||
|
||||
|
||||
@@ -78,7 +78,9 @@ try {
|
||||
$bodyBlock = if ($parts.Count -ge 2) { $parts[1] } else { "{}" }
|
||||
|
||||
$statusLine = ($headersBlock -split "`n" | Select-Object -First 1).Trim()
|
||||
if ($statusLine -notmatch " 20[04-9] ") {
|
||||
# 2xx completo: la creación de repos/hooks/releases responde 201 y la clase
|
||||
# de caracteres anterior ([04-9]) lo trataba como error pese al éxito.
|
||||
if ($statusLine -notmatch " 20[0-9] ") {
|
||||
$snippet = $bodyBlock.Trim()
|
||||
if ($snippet.Length -gt 500) { $snippet = $snippet.Substring(0, 500) + "..." }
|
||||
throw "Gitea API HTTP error: $statusLine`nURI: $uri`nBody: $snippet"
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
Usa `../../agent/SKILL.md`.
|
||||
@@ -1,3 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
Usa `../../agent/TOOLS.md`.
|
||||
@@ -1,6 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
El historial util de incidentes esta consolidado en:
|
||||
|
||||
- `../../docs/runbooks/incidentes-openclaw.md`
|
||||
- `../../docs/runbooks/coolify-docker.md`
|
||||
@@ -1,3 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
Usa `../../agent/SKILL.md`.
|
||||
@@ -1,3 +0,0 @@
|
||||
# Migrado
|
||||
|
||||
Usa `../../agent/TOOLS.md`.
|
||||
@@ -8,18 +8,28 @@
|
||||
# Criterio de exito: psql imprime 3 lineas "UPDATE 1" (una por sentencia).
|
||||
# Tras aplicar, NO pulsar "Refresh" en /super_admin/settings.
|
||||
#
|
||||
# IMPORTANTE: los 3 UPDATE por si solos NO alcanzan cuando el plan ya se habia
|
||||
# revertido a 'community'. Al revertirse, Internal::ReconcilePlanConfigService
|
||||
# apaga los 9 feature flags premium en CADA cuenta, y eso vive en la tabla
|
||||
# accounts (bitmask), no en installation_configs. Usa -ReenableAccountFeatures
|
||||
# para reactivarlos. Ver docs/runbooks/chatwoot-update.md.
|
||||
#
|
||||
# Uso:
|
||||
# .\scripts\Apply-ChatwootEnterprisePatch.ps1
|
||||
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -DryRun
|
||||
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures
|
||||
# .\scripts\Apply-ChatwootEnterprisePatch.ps1 -Container "postgres-c11xzy2tx2cdapm32f5b89vy"
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[switch]$DryRun,
|
||||
[switch]$ReenableAccountFeatures,
|
||||
[string]$ServiceUuid = "c11xzy2tx2cdapm32f5b89vy",
|
||||
[string]$Container = "",
|
||||
[string]$AppContainer = "",
|
||||
[string]$LxcId = "102",
|
||||
[string]$ProxmoxHost = $(if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" }),
|
||||
[string]$SshKey = $(if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" })
|
||||
[string]$SshKey = $(if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { Join-Path $PSScriptRoot "..\keys\proxmox_ed25519" })
|
||||
)
|
||||
|
||||
# Encoding UTF-8 sin BOM (BOM rompe el shebang #!/bin/bash en Linux).
|
||||
@@ -44,13 +54,18 @@ function Invoke-Remote {
|
||||
return & ssh @args
|
||||
}
|
||||
|
||||
if (-not $AppContainer) { $AppContainer = "chatwoot-$ServiceUuid" }
|
||||
|
||||
# --- 1) Resolver contenedor Postgres de Chatwoot --------------------------
|
||||
if (-not $Container) {
|
||||
# Dos greps encadenados en vez de un solo patron: Coolify nombra los
|
||||
# contenedores <servicio>-<uuid> (postgres-c11xzy...), asi que un patron
|
||||
# "<uuid>.*postgres" nunca casa. El orden no importa con greps separados.
|
||||
$detect = @'
|
||||
#!/bin/bash
|
||||
pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -Ei "c11xzy2tx2cdapm32f5b89vy.*(pgvector|postgres|db)" | head -n1'
|
||||
pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -F "__UUID__" | grep -Ei "(pgvector|postgres|db)" | head -n1'
|
||||
'@
|
||||
$detect = $detect.Replace('__LXC__', $LxcId)
|
||||
$detect = $detect.Replace('__LXC__', $LxcId).Replace('__UUID__', $ServiceUuid)
|
||||
|
||||
$tmpDetect = [IO.Path]::GetTempFileName() + ".sh"
|
||||
[IO.File]::WriteAllText($tmpDetect, $detect, $utf8NoBom)
|
||||
@@ -72,8 +87,9 @@ pct exec __LXC__ -- bash -lc 'docker ps --format "{{.Names}}" | grep -Ei "c11xzy
|
||||
if ($LASTEXITCODE -ne 0) { throw "Listado remoto fallo (exit $LASTEXITCODE)." }
|
||||
$Container = @($cand | Where-Object { $_ -match '\S' })[0]
|
||||
if (-not $Container) {
|
||||
throw "No se encontro contenedor. Pasa -Container explicito (ej: postgres-c11xzy2tx2cdapm32f5b89vy)."
|
||||
throw "No se encontro contenedor Postgres para el servicio $ServiceUuid en el LXC $LxcId. Pasa -Container explicito (ej: postgres-$ServiceUuid)."
|
||||
}
|
||||
$Container = $Container.Trim()
|
||||
} finally {
|
||||
Remove-Item -LiteralPath $tmpDetect -ErrorAction SilentlyContinue
|
||||
}
|
||||
@@ -174,8 +190,23 @@ if ($DryRun) {
|
||||
Write-Host "------"
|
||||
Write-Host "[dry-run] apply.sh:"
|
||||
Write-Host "------"
|
||||
Write-Host $applyScript
|
||||
# PGPASSWORD se enmascara: la regla del repo es que ningun secreto salga por
|
||||
# stdout ni quede en un log. (String.Replace revienta con un patron vacio,
|
||||
# de ahi el guard.)
|
||||
$safeApply = if ([string]::IsNullOrEmpty($pgPass)) { $applyScript } else { $applyScript.Replace($pgPass, "********") }
|
||||
Write-Host $safeApply
|
||||
Write-Host "------"
|
||||
if ($ReenableAccountFeatures) {
|
||||
Write-Host "[dry-run] Ademas reactivaria estos feature flags premium en TODAS las cuentas,"
|
||||
Write-Host " via 'rails runner' en $AppContainer :"
|
||||
Write-Host " disable_branding audit_logs sla custom_roles captain_integration"
|
||||
Write-Host " captain_integration_v2 captain_document_auto_sync csat_review_notes"
|
||||
Write-Host " conversation_required_attributes"
|
||||
}
|
||||
else {
|
||||
Write-Host "[dry-run] Los feature flags premium por cuenta NO se tocarian."
|
||||
Write-Host " Agrega -ReenableAccountFeatures si el plan venia de 'community'."
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
@@ -266,3 +297,101 @@ Invoke-Remote "rm -f $remoteSql $remoteApply $remoteVerify" | Out-Null
|
||||
Write-Host ""
|
||||
Write-Host "[OK] Parche enterprise aplicado correctamente (3/3 UPDATE 1)."
|
||||
Write-Host " NO pulsar 'Refresh' en /super_admin/settings."
|
||||
|
||||
# --- 6) Reactivar feature flags premium por cuenta -------------------------
|
||||
# Cuando el plan se revierte a 'community', Internal::ReconcilePlanConfigService
|
||||
# corre account.disable_features!(*premium_features) sobre TODAS las cuentas.
|
||||
# Esos flags viven en accounts.feature_flags (bitmask) y los 3 UPDATE de arriba
|
||||
# no los tocan: hay que reactivarlos explicitamente o la UI sigue sin enterprise.
|
||||
if (-not $ReenableAccountFeatures) {
|
||||
Write-Host ""
|
||||
Write-Host "[!] Los feature flags premium por cuenta NO se tocaron."
|
||||
Write-Host " Si el plan venia de 'community', vuelve a correr con -ReenableAccountFeatures."
|
||||
return
|
||||
}
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "[*] Reactivando feature flags premium por cuenta (arranca Rails, ~40 s) ..."
|
||||
|
||||
$ruby = @'
|
||||
PREMIUM = %w[
|
||||
disable_branding audit_logs sla custom_roles
|
||||
captain_integration captain_integration_v2 captain_document_auto_sync
|
||||
csat_review_notes conversation_required_attributes
|
||||
]
|
||||
# Los 3 UPDATE se hacen por SQL puro, asi que NO disparan el
|
||||
# `after_commit :clear_cache` de InstallationConfig. GlobalConfig cachea en Redis
|
||||
# con TTL de 1 dia (V1:GLOBAL_CONFIG:*), asi que sin esta limpieza la app puede
|
||||
# seguir sirviendo el plan viejo hasta 24 h.
|
||||
GlobalConfig.clear_cache
|
||||
puts "global_config_cache=limpiado"
|
||||
Account.find_each do |account|
|
||||
before = PREMIUM.reject { |f| account.feature_enabled?(f) }
|
||||
account.enable_features!(*PREMIUM)
|
||||
account.reload
|
||||
after = PREMIUM.reject { |f| account.feature_enabled?(f) }
|
||||
puts "account=#{account.id}|#{account.name}|reactivados=#{before.empty? ? 'ninguno' : before.join(',')}|pendientes=#{after.empty? ? 'ninguno' : after.join(',')}"
|
||||
end
|
||||
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
|
||||
'@
|
||||
|
||||
$tmpRb = [IO.Path]::GetTempFileName()
|
||||
$remoteRb = "/tmp/chatwoot-reenable-features.rb"
|
||||
$tmpRunner = [IO.Path]::GetTempFileName()
|
||||
$remoteRunner = "/tmp/chatwoot-reenable-features.sh"
|
||||
|
||||
$runner = @'
|
||||
set -e
|
||||
pct push __LXC__ __RB__ __RB__
|
||||
pct exec __LXC__ -- docker cp __RB__ __APP__:__RB__
|
||||
pct exec __LXC__ -- docker exec -i __APP__ bundle exec rails runner __RB__
|
||||
pct exec __LXC__ -- docker exec -i __APP__ rm -f __RB__
|
||||
pct exec __LXC__ -- rm -f __RB__
|
||||
'@
|
||||
$runner = $runner.Replace('__LXC__', $LxcId).Replace('__APP__', $AppContainer).Replace('__RB__', $remoteRb)
|
||||
|
||||
try {
|
||||
[IO.File]::WriteAllText($tmpRb, $ruby, $utf8NoBom)
|
||||
[IO.File]::WriteAllText($tmpRunner, $runner, $utf8NoBom)
|
||||
|
||||
foreach ($pair in @(@($tmpRb, $remoteRb), @($tmpRunner, $remoteRunner))) {
|
||||
$scpArgs = @(
|
||||
"-o", "BatchMode=yes"
|
||||
"-o", "ConnectTimeout=15"
|
||||
"-o", "StrictHostKeyChecking=no"
|
||||
"-i", $SshKey
|
||||
$pair[0]
|
||||
"root@${ProxmoxHost}:$($pair[1])"
|
||||
)
|
||||
& scp @scpArgs | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) { throw "scp de $($pair[1]) fallo (exit $LASTEXITCODE)." }
|
||||
}
|
||||
|
||||
$featOut = Invoke-Remote "bash $remoteRunner"
|
||||
$featExit = $LASTEXITCODE
|
||||
$featOut | ForEach-Object { Write-Host $_ }
|
||||
}
|
||||
finally {
|
||||
Remove-Item -LiteralPath $tmpRb, $tmpRunner -ErrorAction SilentlyContinue
|
||||
Invoke-Remote "rm -f $remoteRb $remoteRunner" | Out-Null
|
||||
}
|
||||
|
||||
if ($featExit -ne 0) {
|
||||
throw "La reactivacion de feature flags fallo (exit $featExit)."
|
||||
}
|
||||
|
||||
$pending = @($featOut | Where-Object { $_ -match 'pendientes=(?!ninguno)' })
|
||||
if ($pending.Count -gt 0) {
|
||||
throw "Quedaron feature flags premium sin activar en $($pending.Count) cuenta(s). Revisa la salida de arriba."
|
||||
}
|
||||
|
||||
$selfHosted = @($featOut | Where-Object { $_ -like "self_hosted_enterprise=*" })[0]
|
||||
Write-Host ""
|
||||
if ($selfHosted -eq "self_hosted_enterprise=true") {
|
||||
Write-Host "[OK] Enterprise activo: plan=enterprise y feature flags premium reactivados en todas las cuentas."
|
||||
}
|
||||
else {
|
||||
Write-Host "[WARN] Feature flags reactivados, pero ChatwootApp.self_hosted_enterprise? no dio true ($selfHosted)."
|
||||
Write-Host " Reinicia el stack para limpiar el cache de GlobalConfig y vuelve a verificar con:"
|
||||
Write-Host " .\scripts\Get-ChatwootLicenseStatus.ps1 -Deep"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,250 @@
|
||||
# Reporta el estado de la licencia enterprise de Chatwoot en Coolify (LXC 102).
|
||||
#
|
||||
# Solo lectura. Pensado como pre-check y post-check del runbook de actualizacion
|
||||
# (docs/runbooks/chatwoot-update.md).
|
||||
#
|
||||
# Que reporta:
|
||||
# - version instalada vs ultima conocida por el hub
|
||||
# - las 3 filas de public.installation_configs del parche enterprise
|
||||
# - con -Deep: ChatwootApp.self_hosted_enterprise? y los 9 feature flags
|
||||
# premium por cuenta (esto tarda ~40 s porque arranca Rails)
|
||||
#
|
||||
# Uso:
|
||||
# .\scripts\Get-ChatwootLicenseStatus.ps1
|
||||
# .\scripts\Get-ChatwootLicenseStatus.ps1 -Deep
|
||||
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[string]$ServiceUuid = "c11xzy2tx2cdapm32f5b89vy",
|
||||
[string]$AppContainer = "",
|
||||
[string]$DbContainer = "",
|
||||
[switch]$Deep
|
||||
)
|
||||
|
||||
$ErrorActionPreference = "Stop"
|
||||
. (Join-Path $PSScriptRoot "ProxmoxAgent.ps1")
|
||||
|
||||
$config = Assert-ProxmoxConfig
|
||||
$lxc = $config.CoolifyLxc
|
||||
$utf8NoBom = New-Object System.Text.UTF8Encoding($false)
|
||||
|
||||
# Feature flags premium que Internal::ReconcilePlanConfigService apaga cuando el
|
||||
# plan vuelve a 'community' (enterprise/config/premium_features.yml).
|
||||
$premiumFeatures = @(
|
||||
"disable_branding", "audit_logs", "sla", "custom_roles",
|
||||
"captain_integration", "captain_integration_v2", "captain_document_auto_sync",
|
||||
"csat_review_notes", "conversation_required_attributes"
|
||||
)
|
||||
|
||||
function Invoke-Remote {
|
||||
param([string]$RemoteCmd)
|
||||
|
||||
$sshArgs = @(
|
||||
"-o", "BatchMode=yes"
|
||||
"-o", "ConnectTimeout=20"
|
||||
"-o", "StrictHostKeyChecking=no"
|
||||
"-i", $config.SshKey
|
||||
"$($config.User)@$($config.HostName)"
|
||||
$RemoteCmd
|
||||
)
|
||||
return & ssh @sshArgs
|
||||
}
|
||||
|
||||
# Sube un script como archivo y lo ejecuta con bash. Evita el infierno de
|
||||
# escaping de comillas sobre SSH (ver docs/casos/chatwoot-enterprise-patch.md).
|
||||
function Invoke-RemoteScript {
|
||||
param(
|
||||
[string]$Body,
|
||||
[string]$RemoteName
|
||||
)
|
||||
|
||||
$tmp = [IO.Path]::GetTempFileName()
|
||||
try {
|
||||
[IO.File]::WriteAllText($tmp, $Body, $utf8NoBom)
|
||||
$scpArgs = @(
|
||||
"-o", "BatchMode=yes"
|
||||
"-o", "ConnectTimeout=20"
|
||||
"-o", "StrictHostKeyChecking=no"
|
||||
"-i", $config.SshKey
|
||||
$tmp
|
||||
"$($config.User)@$($config.HostName):/tmp/$RemoteName"
|
||||
)
|
||||
& scp @scpArgs | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) { throw "scp de $RemoteName fallo (exit $LASTEXITCODE)." }
|
||||
|
||||
$out = Invoke-Remote "bash /tmp/$RemoteName"
|
||||
$code = $LASTEXITCODE
|
||||
Invoke-Remote "rm -f /tmp/$RemoteName" | Out-Null
|
||||
if ($code -ne 0) { throw "$RemoteName fallo en el host (exit $code)." }
|
||||
return $out
|
||||
}
|
||||
finally {
|
||||
Remove-Item -LiteralPath $tmp -ErrorAction SilentlyContinue
|
||||
}
|
||||
}
|
||||
|
||||
# --- 1) Resolver contenedores del stack -----------------------------------
|
||||
if (-not $AppContainer) { $AppContainer = "chatwoot-$ServiceUuid" }
|
||||
if (-not $DbContainer) { $DbContainer = "postgres-$ServiceUuid" }
|
||||
|
||||
$psScript = @'
|
||||
pct exec __LXC__ -- docker ps --format "{{.Names}} {{.Status}}" | grep -E "__UUID__"
|
||||
'@
|
||||
$psScript = $psScript.Replace('__LXC__', $lxc).Replace('__UUID__', $ServiceUuid)
|
||||
$containers = Invoke-RemoteScript -Body $psScript -RemoteName "cw-status-ps.sh"
|
||||
|
||||
Write-Host "=== Stack Chatwoot (LXC $lxc) ==="
|
||||
if (-not $containers) {
|
||||
throw "No se encontro ningun contenedor con el UUID $ServiceUuid en el LXC $lxc."
|
||||
}
|
||||
$containers | ForEach-Object { Write-Host " $_" }
|
||||
|
||||
# --- 2) Version instalada -------------------------------------------------
|
||||
$verScript = @'
|
||||
pct exec __LXC__ -- docker exec -i __APP__ sh -c 'grep -m1 "version:" /app/config/app.yml'
|
||||
'@
|
||||
$verScript = $verScript.Replace('__LXC__', $lxc).Replace('__APP__', $AppContainer)
|
||||
$verRaw = (Invoke-RemoteScript -Body $verScript -RemoteName "cw-status-ver.sh") -join " "
|
||||
$version = if ($verRaw -match "'([^']+)'") { $Matches[1] } else { $verRaw.Trim() }
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "=== Version ==="
|
||||
Write-Host " instalada: $version"
|
||||
|
||||
# --- 3) Filas del parche enterprise ---------------------------------------
|
||||
# La query se ejecuta dentro del contenedor Postgres para que POSTGRES_USER /
|
||||
# POSTGRES_PASSWORD nunca salgan del contenedor ni queden en logs.
|
||||
$sqlScript = @'
|
||||
pct exec __LXC__ -- docker exec -i __DB__ bash -lc 'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -F "|" -c "SELECT name, serialized_value FROM public.installation_configs ORDER BY name"'
|
||||
'@
|
||||
$sqlScript = $sqlScript.Replace('__LXC__', $lxc).Replace('__DB__', $DbContainer)
|
||||
$rows = Invoke-RemoteScript -Body $sqlScript -RemoteName "cw-status-sql.sh"
|
||||
|
||||
function Get-ConfigValue {
|
||||
param([string]$Name)
|
||||
|
||||
$line = @($rows | Where-Object { $_ -like "$Name|*" })[0]
|
||||
if (-not $line) { return "<ausente>" }
|
||||
# serialized_value es YAML de Ruby: "--- ...\nvalue: X\n"
|
||||
if ($line -match 'value:\s*([^\\"]*)') { return $Matches[1].Trim() }
|
||||
return $line
|
||||
}
|
||||
|
||||
$plan = Get-ConfigValue "INSTALLATION_PRICING_PLAN"
|
||||
$quantity = Get-ConfigValue "INSTALLATION_PRICING_PLAN_QUANTITY"
|
||||
$identifier = Get-ConfigValue "INSTALLATION_IDENTIFIER"
|
||||
$instName = Get-ConfigValue "INSTALLATION_NAME"
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "=== installation_configs (parche enterprise) ==="
|
||||
Write-Host " INSTALLATION_PRICING_PLAN = $plan"
|
||||
Write-Host " INSTALLATION_PRICING_PLAN_QUANTITY = $quantity"
|
||||
Write-Host " INSTALLATION_IDENTIFIER = $identifier"
|
||||
Write-Host " INSTALLATION_NAME = $instName"
|
||||
|
||||
# La ventana diaria de revert es determinista:
|
||||
# Internal::TriggerDailyScheduledItemsJob (cron 0 0 * * *) programa
|
||||
# Internal::CheckNewVersionsJob en beginning_of_day + (MD5(identifier).hex % 1440) minutos.
|
||||
if ($identifier -and $identifier -ne "<ausente>") {
|
||||
$md5 = [System.Security.Cryptography.MD5]::Create()
|
||||
try {
|
||||
$digest = ($md5.ComputeHash([Text.Encoding]::UTF8.GetBytes($identifier)) |
|
||||
ForEach-Object { $_.ToString("x2") }) -join ""
|
||||
$asInt = [Numerics.BigInteger]::Parse("0$digest", "AllowHexSpecifier")
|
||||
$minute = [int]($asInt % 1440)
|
||||
Write-Host (" ventana diaria de revert = {0:d2}:{1:d2} UTC (minuto {2})" -f [int][math]::Floor($minute / 60), [int]($minute % 60), $minute)
|
||||
}
|
||||
finally {
|
||||
$md5.Dispose()
|
||||
}
|
||||
}
|
||||
|
||||
$planOk = ($plan -eq "enterprise")
|
||||
|
||||
# --- 4) Chequeo profundo con Rails ----------------------------------------
|
||||
$featuresOk = $null
|
||||
if ($Deep) {
|
||||
Write-Host ""
|
||||
Write-Host "[*] Arrancando Rails para el chequeo profundo (~40 s) ..."
|
||||
|
||||
$ruby = @'
|
||||
PREMIUM = %w[__FEATURES__]
|
||||
puts "version=#{Chatwoot.config[:version]}"
|
||||
puts "enterprise=#{ChatwootApp.enterprise?}"
|
||||
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
|
||||
puts "latest_known_version=#{Redis::Alfred.get(Redis::Alfred::LATEST_CHATWOOT_VERSION)}"
|
||||
Account.find_each do |a|
|
||||
off = PREMIUM.reject { |f| a.feature_enabled?(f) }
|
||||
puts "account=#{a.id}|#{a.name}|#{off.empty? ? 'OK' : off.join(',')}"
|
||||
end
|
||||
'@
|
||||
$ruby = $ruby.Replace('__FEATURES__', ($premiumFeatures -join " "))
|
||||
|
||||
$tmpRb = [IO.Path]::GetTempFileName()
|
||||
try {
|
||||
[IO.File]::WriteAllText($tmpRb, $ruby, $utf8NoBom)
|
||||
$scpArgs = @(
|
||||
"-o", "BatchMode=yes"
|
||||
"-o", "ConnectTimeout=20"
|
||||
"-o", "StrictHostKeyChecking=no"
|
||||
"-i", $config.SshKey
|
||||
$tmpRb
|
||||
"$($config.User)@$($config.HostName):/tmp/cw-deep.rb"
|
||||
)
|
||||
& scp @scpArgs | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) { throw "scp del script Ruby fallo (exit $LASTEXITCODE)." }
|
||||
|
||||
# El .rb tiene que llegar al filesystem del contenedor: host -> LXC -> docker.
|
||||
$runner = @'
|
||||
set -e
|
||||
pct push __LXC__ /tmp/cw-deep.rb /tmp/cw-deep.rb
|
||||
pct exec __LXC__ -- docker cp /tmp/cw-deep.rb __APP__:/tmp/cw-deep.rb
|
||||
pct exec __LXC__ -- docker exec -i __APP__ bundle exec rails runner /tmp/cw-deep.rb
|
||||
pct exec __LXC__ -- docker exec -i __APP__ rm -f /tmp/cw-deep.rb
|
||||
pct exec __LXC__ -- rm -f /tmp/cw-deep.rb
|
||||
'@
|
||||
$runner = $runner.Replace('__LXC__', $lxc).Replace('__APP__', $AppContainer)
|
||||
$deepOut = Invoke-RemoteScript -Body $runner -RemoteName "cw-status-deep.sh"
|
||||
Invoke-Remote "rm -f /tmp/cw-deep.rb" | Out-Null
|
||||
}
|
||||
finally {
|
||||
Remove-Item -LiteralPath $tmpRb -ErrorAction SilentlyContinue
|
||||
}
|
||||
|
||||
$selfHosted = @($deepOut | Where-Object { $_ -like "self_hosted_enterprise=*" })[0]
|
||||
$latest = @($deepOut | Where-Object { $_ -like "latest_known_version=*" })[0]
|
||||
$accounts = @($deepOut | Where-Object { $_ -like "account=*" })
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "=== Rails ==="
|
||||
if ($latest) { Write-Host " $latest" }
|
||||
if ($selfHosted) { Write-Host " $selfHosted" }
|
||||
|
||||
Write-Host ""
|
||||
Write-Host "=== Feature flags premium por cuenta ==="
|
||||
$featuresOk = $true
|
||||
foreach ($line in $accounts) {
|
||||
$parts = $line.Substring(8) -split '\|', 3
|
||||
$state = if ($parts.Count -ge 3) { $parts[2] } else { "?" }
|
||||
if ($state -ne "OK") { $featuresOk = $false }
|
||||
Write-Host (" cuenta {0} ({1}): {2}" -f $parts[0], $parts[1], $(if ($state -eq "OK") { "todos activos" } else { "APAGADOS -> $state" }))
|
||||
}
|
||||
}
|
||||
|
||||
# --- 5) Veredicto ---------------------------------------------------------
|
||||
Write-Host ""
|
||||
if ($planOk -and ($featuresOk -eq $true)) {
|
||||
Write-Host "[OK] Enterprise activo: plan=enterprise y feature flags premium completos."
|
||||
}
|
||||
elseif ($planOk -and ($null -eq $featuresOk)) {
|
||||
Write-Host "[OK] plan=enterprise. Corre con -Deep para confirmar los feature flags por cuenta."
|
||||
}
|
||||
elseif ($planOk) {
|
||||
Write-Host "[WARN] plan=enterprise pero hay feature flags premium apagados."
|
||||
Write-Host " Corre: .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures"
|
||||
}
|
||||
else {
|
||||
Write-Host "[FAIL] Enterprise NO activo (plan=$plan, quantity=$quantity)."
|
||||
Write-Host " Corre: .\scripts\Apply-ChatwootEnterprisePatch.ps1 -ReenableAccountFeatures"
|
||||
Write-Host " Detalle del caso: docs/runbooks/chatwoot-update.md"
|
||||
}
|
||||
@@ -96,12 +96,20 @@ if ($VerifyOnly) {
|
||||
echo '--- onboot / startup on LXC $lxc ---'
|
||||
grep -Ei 'onboot|startup' /etc/pve/lxc/$lxc.conf 2>/dev/null || echo '(onboot not set -> defaults to 0, will NOT auto-start)'
|
||||
echo '--- guardian unit ---'
|
||||
systemctl is-enabled $svcName 2>/dev/null || echo '($svcName not installed/enabled)'
|
||||
printf 'enabled: '; systemctl is-enabled $svcName 2>/dev/null || echo '(not installed/enabled)'
|
||||
printf 'state: '; systemctl is-active $svcName 2>/dev/null || true
|
||||
printf 'result: '; systemctl show $svcName -p Result --value 2>/dev/null || true
|
||||
printf 'timeout: '; systemctl show $svcName -p TimeoutStartUSec --value 2>/dev/null || true
|
||||
echo '--- last boot outcome (journal) ---'
|
||||
journalctl -u $svcName --no-pager -b 2>/dev/null | tail -n 6 || echo '(no journal for this boot)'
|
||||
echo '--- guardian script present? ---'
|
||||
test -x $binPath && echo 'present ($binPath)' || echo 'missing ($binPath)'
|
||||
echo '--- last log lines ---'
|
||||
tail -n 15 $logPath 2>/dev/null || echo '(no log yet)'
|
||||
"@
|
||||
Write-Host ""
|
||||
Write-Host "Healthy = enabled:enabled / state:active / result:success and a log run ending in 'done'." -ForegroundColor DarkGray
|
||||
Write-Host "state:failed or a log run that stops after 'LXC ... running' means the guardian died mid-wait." -ForegroundColor DarkGray
|
||||
return
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
param(
|
||||
[Parameter(Mandatory = $true)]
|
||||
[string]$BashCommand
|
||||
)
|
||||
|
||||
$ErrorActionPreference = "Stop"
|
||||
. "$PSScriptRoot\ProxmoxAgent.ps1"
|
||||
|
||||
$text = $BashCommand -replace "`r`n", "`n"
|
||||
$bytes = [System.Text.Encoding]::UTF8.GetBytes($text)
|
||||
$b64 = [Convert]::ToBase64String($bytes)
|
||||
|
||||
# On Proxmox host, run pct exec with base64 decoded directly inside container
|
||||
$remoteCmd = "pct exec 102 -- bash -c 'echo " + $b64 + " | base64 -d | bash'"
|
||||
Invoke-ProxmoxSshCommand -Command $remoteCmd
|
||||
@@ -10,7 +10,7 @@ function Get-ProxmoxConfig {
|
||||
HostName = if ($env:PROXMOX_HOST) { $env:PROXMOX_HOST } else { "192.168.0.200" }
|
||||
Node = if ($env:PROXMOX_NODE) { $env:PROXMOX_NODE } else { "thinkcentre" }
|
||||
User = if ($env:PROXMOX_USER) { $env:PROXMOX_USER } else { "root" }
|
||||
SshKey = if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win" }
|
||||
SshKey = if ($env:PROXMOX_SSH_KEY) { $env:PROXMOX_SSH_KEY } else { Join-Path $PSScriptRoot "..\keys\proxmox_ed25519" }
|
||||
ApiBaseUrl = if ($env:PROXMOX_API_BASE_URL) { $env:PROXMOX_API_BASE_URL } else { "https://192.168.0.200:8006/api2/json" }
|
||||
ApiTokenHeader = $tokenHeader
|
||||
CoolifyLxc = if ($env:PROXMOX_COOLIFY_LXC) { $env:PROXMOX_COOLIFY_LXC } else { "102" }
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
$env:PROXMOX_HOST = "192.168.0.200"
|
||||
$env:PROXMOX_NODE = "thinkcentre"
|
||||
$env:PROXMOX_USER = "root"
|
||||
$env:PROXMOX_SSH_KEY = "C:\Users\Uriel Jareth\.openclaw\workspace\proxmox_key_win"
|
||||
$env:PROXMOX_SSH_KEY = "keys\proxmox_ed25519" # relativo a la raíz del repo (≡ ~\.ssh\coolify_key)
|
||||
$env:PROXMOX_API_BASE_URL = "https://192.168.0.200:8006/api2/json"
|
||||
$env:PROXMOX_API_TOKEN_ID = "root@pam!openclaw"
|
||||
$env:PROXMOX_API_TOKEN_SECRET = "REPLACE_WITH_TOKEN_SECRET"
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
# ===========================================================================
|
||||
# Deploy-OhDaddy.ps1
|
||||
# Redeploy de oh-daddy (https://github.com/KenKaiii/oh-daddy) en Coolify
|
||||
# (LXC 102) SIN depender del pull de Coolify (la imagen es local:
|
||||
# oh-daddy-app:local, construida en el server - patron Deploy-SoloLeveling).
|
||||
#
|
||||
# Stack: app (Next.js 16) + db (postgres:17) + inngest (self-hosted v1.44.0)
|
||||
# + inngest-db + inngest-redis. Servicio Coolify rzittzudkunwx8gilonn7tqe,
|
||||
# proyecto "AI AGENCY" / production. Dominio: ohdaddy.urieljareth.org.
|
||||
#
|
||||
# Pre-requisitos:
|
||||
# - .env.local.ps1 con COOLIFY_*, PROXMOX_* (los secretos OH_DADDY_* solo
|
||||
# se necesitan si vas a re-aplicar envs; el deploy los lee del .env del
|
||||
# servicio en disco).
|
||||
# - stacks/oh-daddy/{Dockerfile,.dockerignore} en este repo (se inyectan
|
||||
# al clone via base64).
|
||||
#
|
||||
# Uso:
|
||||
# . .\.env.local.ps1
|
||||
# .\scripts\apps\Deploy-OhDaddy.ps1 # build + up + schema + registro
|
||||
# .\scripts\apps\Deploy-OhDaddy.ps1 -NoBuild # solo up + verificaciones
|
||||
# ===========================================================================
|
||||
[CmdletBinding()]
|
||||
param(
|
||||
[string]$ServiceUuid = 'rzittzudkunwx8gilonn7tqe',
|
||||
[string]$Fqdn = 'https://ohdaddy.urieljareth.org',
|
||||
[string]$Repo = 'https://github.com/KenKaiii/oh-daddy.git',
|
||||
[string]$Image = 'oh-daddy-app:local',
|
||||
[switch]$NoBuild
|
||||
)
|
||||
Set-Location 'H:\MegaSync\Proyectos\Proxmox & Coolify Manager'
|
||||
. .\.env.local.ps1
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
function Invoke-OnServer([string]$scriptBody, [int]$TimeoutSec = 600) {
|
||||
$body = ($scriptBody -replace "`r`n","`n") -replace "`r","`n"
|
||||
$b64 = [Convert]::ToBase64String([Text.Encoding]::UTF8.GetBytes($body))
|
||||
$cmd = "pct exec 102 -- sh -c 'echo $b64 | base64 -d | sh'"
|
||||
& .\scripts\Invoke-ProxmoxSsh.ps1 -Command $cmd
|
||||
}
|
||||
|
||||
$stackDir = 'H:\MegaSync\Proyectos\Proxmox & Coolify Manager\stacks\oh-daddy'
|
||||
|
||||
# ---------------------------------------------------------------- 1. BUILD ---
|
||||
if (-not $NoBuild) {
|
||||
Write-Host "==> Construyendo imagen en el server (clone + docker build)..." -ForegroundColor Cyan
|
||||
$dfB64 = [Convert]::ToBase64String([IO.File]::ReadAllBytes((Join-Path $stackDir 'Dockerfile')))
|
||||
$diB64 = [Convert]::ToBase64String([IO.File]::ReadAllBytes((Join-Path $stackDir '.dockerignore')))
|
||||
$build = @"
|
||||
set -e
|
||||
BUILD=/tmp/oh-daddy-build
|
||||
rm -rf "`$BUILD" /tmp/oh-daddy-build.log
|
||||
mkdir -p "`$BUILD"; cd "`$BUILD"
|
||||
git clone --depth 1 $Repo repo
|
||||
cd repo
|
||||
echo "HEAD: `$(git log -1 --oneline)"
|
||||
echo '$dfB64' | base64 -d > Dockerfile
|
||||
echo '$diB64' | base64 -d > .dockerignore
|
||||
nohup sh -c 'cd /tmp/oh-daddy-build/repo && docker build --build-arg NEXT_PUBLIC_APP_URL=$Fqdn -t $Image . ; echo BUILD_EXIT=`$?' > /tmp/oh-daddy-build.log 2>&1 &
|
||||
echo BUILD_STARTED
|
||||
"@
|
||||
Invoke-OnServer $build
|
||||
Write-Host " (build en background; log: pct exec 102 -- tail -f /tmp/oh-daddy-build.log)" -ForegroundColor DarkGray
|
||||
Write-Host "==> Esperando build (poll cada 30s, max 15min)..." -ForegroundColor Cyan
|
||||
$deadline = (Get-Date).AddMinutes(15)
|
||||
while ((Get-Date) -lt $deadline) {
|
||||
Start-Sleep -Seconds 30
|
||||
$st = Invoke-OnServer "set +e`ntail -3 /tmp/oh-daddy-build.log`ndocker images $Image --format '{{.ID}}'"
|
||||
if ($st -match 'BUILD_EXIT=0') { Write-Host "OK: imagen $Image construida." -ForegroundColor Green; break }
|
||||
if ($st -match 'BUILD_EXIT=([1-9][0-9]*)') { throw "Build fallo (exit $($Matches[1])). Log: /tmp/oh-daddy-build.log" }
|
||||
Write-Host " ... compilando" -ForegroundColor DarkGray
|
||||
}
|
||||
if (-not $st) { throw "No se pudo confirmar el build." }
|
||||
}
|
||||
|
||||
# -------------------------------------------------- 2. UP + PROXY + SCHEMA ---
|
||||
Write-Host "==> Levantando stack, conectando proxy y aplicando schema..." -ForegroundColor Cyan
|
||||
$up = @"
|
||||
set -e
|
||||
SVC=/data/coolify/services/$ServiceUuid
|
||||
cd "`$SVC"
|
||||
docker compose up -d
|
||||
docker network connect $ServiceUuid coolify-proxy 2>&1 && echo PROXY_CONNECTED || echo "(proxy ya conectado)"
|
||||
echo "=== SCHEMA (idempotente) ==="
|
||||
docker exec -i db-$ServiceUuid psql -U ohdaddy -d ohdaddy < /tmp/oh-daddy-build/repo/db/schema.sql 2>&1 | grep -cE 'ERROR' | sed 's/^/errores_sql=/' || true
|
||||
echo "=== PS ==="
|
||||
sleep 15
|
||||
docker compose ps --format '{{.Name}} | {{.Status}}'
|
||||
"@
|
||||
Invoke-OnServer $up
|
||||
|
||||
# ------------------------------------------- 3. REGISTRO INNGEST + SALUD ---
|
||||
Write-Host "==> Re-registrando funciones en Inngest (PUT publico)..." -ForegroundColor Cyan
|
||||
$code = & curl.exe -s -o NUL -w "%{http_code}" --max-time 60 -X PUT "$Fqdn/api/inngest"
|
||||
Write-Host ("INNGEST_REGISTER=" + $code)
|
||||
if ($code -ne '200') { Write-Host "AVISO: registro no confirmado. Reintentar cuando la app este healthy: curl -X PUT $Fqdn/api/inngest" -ForegroundColor Yellow }
|
||||
|
||||
Write-Host "=== HEALTH publico: $Fqdn ===" -ForegroundColor Cyan
|
||||
$code2 = & curl.exe -s -o NUL -w "%{http_code}" --max-time 30 "$Fqdn/login"
|
||||
Write-Host ("PUBLIC_LOGIN=" + $code2)
|
||||
if ($code2 -eq '200') { Write-Host "OK: oh-daddy activo en $Fqdn" -ForegroundColor Green }
|
||||
else { Write-Host "AVISO: /login no responde 200 ($code2). Logs: docker logs app-$ServiceUuid" -ForegroundColor Yellow }
|
||||
@@ -0,0 +1,140 @@
|
||||
#!/bin/bash
|
||||
# Guard: mantiene la edicion enterprise de Chatwoot en LXC 102.
|
||||
#
|
||||
# Por que existe: Internal::CheckNewVersionsJob hace ping diario a
|
||||
# hub.2.chatwoot.com (a las 16:16 UTC para este installation_identifier) y
|
||||
# reescribe INSTALLATION_PRICING_PLAN con lo que responda el hub ('community'),
|
||||
# y ademas Internal::ReconcilePlanConfigService apaga los 9 feature flags
|
||||
# premium en TODAS las cuentas.
|
||||
#
|
||||
# No se bloquea el hub a proposito: ese mismo host relaya las notificaciones
|
||||
# push del movil (ChatwootHub.send_push) porque FIREBASE_* esta vacio. Bloquearlo
|
||||
# romperia el push. En vez de eso, este guard detecta el revert y lo deshace.
|
||||
#
|
||||
# Cheap por diseno: en el caso normal hace 1 SELECT y sale. Solo cuando detecta
|
||||
# plan != enterprise levanta Rails para limpiar cache y reactivar flags.
|
||||
#
|
||||
# Instalado por el runbook docs/runbooks/chatwoot-update.md
|
||||
# Cron: */15 * * * *
|
||||
|
||||
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
LXC=102
|
||||
UUID=c11xzy2tx2cdapm32f5b89vy
|
||||
DB_CT="postgres-$UUID"
|
||||
APP_CT="chatwoot-$UUID"
|
||||
LOG=/var/log/chatwoot-enterprise-guard.log
|
||||
LOCK=/var/lock/chatwoot-enterprise-guard.lock
|
||||
MAX_LOG=2097152
|
||||
|
||||
log() { echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $*" >> "$LOG"; }
|
||||
|
||||
# Rotacion simple para que el log no crezca sin control.
|
||||
if [ -f "$LOG" ] && [ "$(stat -c %s "$LOG" 2>/dev/null || echo 0)" -gt "$MAX_LOG" ]; then
|
||||
mv -f "$LOG" "$LOG.1"
|
||||
fi
|
||||
|
||||
# Una sola instancia a la vez (el camino de reparacion tarda ~1 min).
|
||||
exec 9>"$LOCK" || exit 0
|
||||
flock -n 9 || exit 0
|
||||
|
||||
# --- 1) Chequeo barato: en que plan estamos ---------------------------------
|
||||
# Se traen todas las filas y se filtra con grep a proposito: un WHERE con
|
||||
# literales necesitaria comillas simples anidadas dentro de bash -lc '...' y eso
|
||||
# es exactamente la clase de escaping que rompe estos scripts.
|
||||
PLAN=$(pct exec "$LXC" -- docker exec -i "$DB_CT" bash -lc \
|
||||
'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -At -F "|" -c "SELECT name, serialized_value::text FROM public.installation_configs"' \
|
||||
2>/dev/null | grep '^INSTALLATION_PRICING_PLAN|')
|
||||
|
||||
if [ -z "$PLAN" ]; then
|
||||
log "ERROR: no se pudo leer INSTALLATION_PRICING_PLAN (stack caido o contenedor renombrado?)"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
case "$PLAN" in
|
||||
*enterprise*)
|
||||
# Caso normal: nada que hacer, y no ensuciamos el log.
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
|
||||
log "DETECTADO revert -> plan actual: $PLAN . Reparando..."
|
||||
|
||||
# --- 2) Reponer las 3 filas del parche --------------------------------------
|
||||
cat > /tmp/cw-guard.sql <<'SQLEOF'
|
||||
UPDATE public.installation_configs
|
||||
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: enterprise\n"'
|
||||
WHERE name = 'INSTALLATION_PRICING_PLAN';
|
||||
|
||||
UPDATE public.installation_configs
|
||||
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: 10000\n"'
|
||||
WHERE name = 'INSTALLATION_PRICING_PLAN_QUANTITY';
|
||||
|
||||
UPDATE public.installation_configs
|
||||
SET serialized_value = '"--- !ruby/hash:ActiveSupport::HashWithIndifferentAccess\nvalue: e04t63ee-5gg8-4b94-8914-ed8137a7d938\n"'
|
||||
WHERE name = 'INSTALLATION_IDENTIFIER';
|
||||
SQLEOF
|
||||
|
||||
SQL_OUT=$(pct exec "$LXC" -- docker exec -i "$DB_CT" bash -lc \
|
||||
'PGPASSWORD="$POSTGRES_PASSWORD" psql -U "$POSTGRES_USER" -d "$POSTGRES_DB" -v ON_ERROR_STOP=1' \
|
||||
< /tmp/cw-guard.sql 2>&1)
|
||||
SQL_RC=$?
|
||||
rm -f /tmp/cw-guard.sql
|
||||
|
||||
UPDATES=$(printf '%s\n' "$SQL_OUT" | grep -c '^UPDATE 1$')
|
||||
if [ "$SQL_RC" -ne 0 ] || [ "$UPDATES" -ne 3 ]; then
|
||||
log "ERROR: los UPDATE fallaron (rc=$SQL_RC, 'UPDATE 1'=$UPDATES). Salida: $SQL_OUT"
|
||||
exit 1
|
||||
fi
|
||||
log "OK: 3/3 UPDATE aplicados"
|
||||
|
||||
# --- 3) Limpiar cache de GlobalConfig y reactivar flags premium -------------
|
||||
# El UPDATE por SQL no dispara el after_commit :clear_cache de InstallationConfig,
|
||||
# y GlobalConfig cachea en Redis con TTL de 1 dia: sin este paso la app puede
|
||||
# seguir sirviendo 'community'. Ademas hay que reactivar los flags por cuenta,
|
||||
# que viven en accounts.feature_flags (bitmask) y el SQL de arriba no toca.
|
||||
cat > /tmp/cw-guard.rb <<'RBEOF'
|
||||
PREMIUM = %w[
|
||||
disable_branding audit_logs sla custom_roles
|
||||
captain_integration captain_integration_v2 captain_document_auto_sync
|
||||
csat_review_notes conversation_required_attributes
|
||||
]
|
||||
GlobalConfig.clear_cache
|
||||
Account.find_each do |account|
|
||||
account.enable_features!(*PREMIUM)
|
||||
account.reload
|
||||
pend = PREMIUM.reject { |f| account.feature_enabled?(f) }
|
||||
puts "account=#{account.id} pendientes=#{pend.empty? ? 'ninguno' : pend.join(',')}"
|
||||
end
|
||||
puts "self_hosted_enterprise=#{ChatwootApp.self_hosted_enterprise?}"
|
||||
RBEOF
|
||||
|
||||
pct push "$LXC" /tmp/cw-guard.rb /tmp/cw-guard.rb
|
||||
pct exec "$LXC" -- docker cp /tmp/cw-guard.rb "$APP_CT":/tmp/cw-guard.rb
|
||||
RB_OUT=$(pct exec "$LXC" -- docker exec -i "$APP_CT" bundle exec rails runner /tmp/cw-guard.rb 2>&1)
|
||||
RB_RC=$?
|
||||
pct exec "$LXC" -- docker exec -i "$APP_CT" rm -f /tmp/cw-guard.rb
|
||||
pct exec "$LXC" -- rm -f /tmp/cw-guard.rb
|
||||
rm -f /tmp/cw-guard.rb
|
||||
|
||||
RESULT=$(printf '%s\n' "$RB_OUT" | grep -E '^(account=|self_hosted_enterprise=)' | tr '\n' ' ')
|
||||
if [ "$RB_RC" -ne 0 ]; then
|
||||
log "ERROR: rails runner fallo (rc=$RB_RC). Salida: $RB_OUT"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
case "$RESULT" in
|
||||
*"self_hosted_enterprise=true"*)
|
||||
case "$RESULT" in
|
||||
*"pendientes=ninguno"*)
|
||||
log "REPARADO: $RESULT"
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
log "PARCIAL: enterprise activo pero quedaron flags pendientes -> $RESULT"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
log "ERROR: tras reparar, self_hosted_enterprise no dio true -> $RESULT"
|
||||
exit 1
|
||||
@@ -6,12 +6,20 @@ Description=Auto-start Coolify LXC + Docker stack + Cloudflare tunnel after boot
|
||||
# policies, not a replacement for them.
|
||||
After=pve-guests.service network-online.target
|
||||
Wants=pve-guests.service network-online.target
|
||||
# Bound retries so a genuinely broken stack doesn't loop forever.
|
||||
StartLimitIntervalSec=3600
|
||||
StartLimitBurst=3
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/bin/coolify-autostart.sh
|
||||
RemainAfterExit=yes
|
||||
TimeoutStartSec=300
|
||||
# Must stay above the script's own budget (COOLIFY_MAX_WAIT 600 + COOLIFY_SETTLE
|
||||
# 180 + step overhead). The old 300 s killed the guardian mid-wait on every real
|
||||
# boot: the `coolify` container only starts ~7m40s after power-on.
|
||||
TimeoutStartSec=1200
|
||||
Restart=on-failure
|
||||
RestartSec=60
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
||||
@@ -8,20 +8,40 @@
|
||||
#
|
||||
# Installed by scripts/Install-CoolifyAutostart.ps1 to /usr/local/bin/ and
|
||||
# invoked by the systemd unit coolify-autostart.service on multi-user.target.
|
||||
#
|
||||
# Timing note (measured on the 2026-08-07 boots): pve-guests takes ~78 s to
|
||||
# start CT 102, the Docker daemon inside it only answers ~4-5 min after boot,
|
||||
# and the last core container (`coolify`) starts ~7m40s after boot. Every wait
|
||||
# here is therefore wall-clock based and generously sized; the systemd unit's
|
||||
# TimeoutStartSec must stay above MAX_WAIT + SETTLE.
|
||||
# -----------------------------------------------------------------------------
|
||||
set -uo pipefail
|
||||
export PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
LXC_ID="${COOLIFY_LXC:-102}"
|
||||
LOG="/var/log/coolify-autostart.log"
|
||||
MAX_WAIT="${COOLIFY_MAX_WAIT:-180}" # seconds to wait for docker inside the LXC
|
||||
LOG="${COOLIFY_LOG:-/var/log/coolify-autostart.log}" # override to dry-run without touching the real log
|
||||
MAX_WAIT="${COOLIFY_MAX_WAIT:-600}" # wall-clock seconds to wait for docker
|
||||
SETTLE="${COOLIFY_SETTLE:-180}" # grace for docker to start its own containers
|
||||
PROBE_TIMEOUT="${COOLIFY_PROBE_TIMEOUT:-20}" # hard timeout for every call into the LXC
|
||||
|
||||
# Dependencies first, then the app, proxy and tunnel. These all normally come
|
||||
# up on their own via Docker restart policies; this loop only heals the ones
|
||||
# that didn't (e.g. restart=no, or a wedged start).
|
||||
CORE_CONTAINERS=(coolify-db coolify-redis coolify-realtime coolify coolify-proxy cloudflared)
|
||||
|
||||
failures=0
|
||||
|
||||
log() { echo "[$(date '+%F %T')] $*" | tee -a "$LOG"; }
|
||||
now() { date +%s; }
|
||||
|
||||
# Every call into the LXC gets a hard timeout. During a cold boot the Docker
|
||||
# daemon is busy starting ~50 containers and a single blocking `docker` call
|
||||
# used to eat the entire budget silently, so the guardian was killed by systemd
|
||||
# before it ever reached the container checks.
|
||||
lxc() { timeout "$PROBE_TIMEOUT" pct exec "$LXC_ID" -- "$@"; }
|
||||
|
||||
# true | false | unknown (unknown = inspect failed, container may not exist yet)
|
||||
container_state() { lxc docker inspect -f '{{.State.Running}}' "$1" 2>/dev/null || echo unknown; }
|
||||
|
||||
log "=== coolify-autostart start (LXC ${LXC_ID}) ==="
|
||||
|
||||
@@ -33,35 +53,48 @@ if [ "$status" != "running" ]; then
|
||||
log "pct start issued"
|
||||
else
|
||||
log "ERROR: pct start ${LXC_ID} failed"
|
||||
failures=$((failures + 1))
|
||||
fi
|
||||
else
|
||||
log "LXC ${LXC_ID} already running"
|
||||
fi
|
||||
|
||||
# 2. Wait for the Docker daemon inside the LXC to respond ----------------------
|
||||
waited=0
|
||||
until pct exec "$LXC_ID" -- docker info >/dev/null 2>&1; do
|
||||
if [ "$waited" -ge "$MAX_WAIT" ]; then
|
||||
log "ERROR: docker not ready after ${MAX_WAIT}s -> aborting"
|
||||
# Wall-clock deadline (not a sleep counter) and a cheap probe: `docker
|
||||
# version` hits /version, while `docker info` enumerates every container and
|
||||
# plugin and stalls for minutes on a loaded daemon.
|
||||
start_ts="$(now)"
|
||||
deadline=$((start_ts + MAX_WAIT))
|
||||
until lxc docker version --format '{{.Server.Version}}' >/dev/null 2>&1; do
|
||||
if [ "$(now)" -ge "$deadline" ]; then
|
||||
log "ERROR: docker not ready after ${MAX_WAIT}s (wall clock) -> aborting"
|
||||
exit 1
|
||||
fi
|
||||
sleep 5
|
||||
waited=$((waited + 5))
|
||||
done
|
||||
log "docker ready after ${waited}s"
|
||||
log "docker ready after $(( $(now) - start_ts ))s"
|
||||
|
||||
# 3. Ensure the core Coolify containers + tunnel container are running ---------
|
||||
# Docker's own restart policies bring these up over several minutes, so poll
|
||||
# each one until SETTLE expires before forcing a start.
|
||||
settle_deadline=$(($(now) + SETTLE))
|
||||
for c in "${CORE_CONTAINERS[@]}"; do
|
||||
st="$(pct exec "$LXC_ID" -- docker inspect -f '{{.State.Running}}' "$c" 2>/dev/null || echo missing)"
|
||||
st="$(container_state "$c")"
|
||||
while [ "$st" != "true" ] && [ "$(now)" -lt "$settle_deadline" ]; do
|
||||
sleep 5
|
||||
st="$(container_state "$c")"
|
||||
done
|
||||
case "$st" in
|
||||
true) log "container ${c}: running" ;;
|
||||
missing) log "WARN container ${c}: not found (skipping)" ;;
|
||||
*)
|
||||
log "container ${c}: running=${st} -> starting"
|
||||
if pct exec "$LXC_ID" -- docker start "$c" >/dev/null 2>&1; then
|
||||
if lxc docker start "$c" >/dev/null 2>&1; then
|
||||
log "started ${c}"
|
||||
elif [ "$st" = "unknown" ]; then
|
||||
log "WARN container ${c}: not found (skipping)"
|
||||
else
|
||||
log "ERROR: could not start ${c}"
|
||||
failures=$((failures + 1))
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
@@ -70,12 +103,24 @@ done
|
||||
# 4. Ensure the systemd cloudflared tunnel inside the LXC is up ----------------
|
||||
# (second connector to the same tunnel; belt-and-suspenders alongside the
|
||||
# cloudflared Docker container above.)
|
||||
if pct exec "$LXC_ID" -- systemctl is-enabled cloudflared >/dev/null 2>&1; then
|
||||
if pct exec "$LXC_ID" -- systemctl start cloudflared >/dev/null 2>&1; then
|
||||
log "cloudflared.service ensured up"
|
||||
else
|
||||
if lxc systemctl is-enabled cloudflared >/dev/null 2>&1; then
|
||||
if ! lxc systemctl start cloudflared >/dev/null 2>&1; then
|
||||
log "WARN: cloudflared.service start returned non-zero"
|
||||
fi
|
||||
log "cloudflared.service is-active=$(lxc systemctl is-active cloudflared 2>/dev/null || echo unknown)"
|
||||
else
|
||||
log "WARN: cloudflared.service not enabled inside LXC ${LXC_ID}"
|
||||
fi
|
||||
|
||||
log "=== coolify-autostart done ==="
|
||||
# 5. Prove the tunnel actually reached Cloudflare this boot --------------------
|
||||
# Without this the log said "ensured up" even when no connector registered.
|
||||
conns="$(lxc journalctl -u cloudflared -b --no-pager 2>/dev/null | grep -c 'Registered tunnel connection' || true)"
|
||||
conns="${conns//[!0-9]/}" # grep -c exits 1 on zero matches; keep only the digits
|
||||
conns="${conns:-0}"
|
||||
log "cloudflared.service registered tunnel connections this boot: ${conns}"
|
||||
if [ "$conns" -eq 0 ]; then
|
||||
log "WARN: no tunnel connection registered yet (edge may still be connecting)"
|
||||
fi
|
||||
|
||||
log "=== coolify-autostart done (failures=${failures}) ==="
|
||||
[ "$failures" -eq 0 ] || exit 1
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
# Evolution Go (API WhatsApp en Go) — stack para el host casero.
|
||||
#
|
||||
# Fuente: https://github.com/evolution-foundation/evolution-go (clon local en
|
||||
# projects/evolution-go, tag 0.7.2). Imagen publicada: evoapicloud/evolution-go.
|
||||
#
|
||||
# Por qué este compose y no el de upstream (docker/examples/docker-compose.yml):
|
||||
# - publica `ports:` 4000 y 5432 al host; aquí el 80/443 y el resto de puertos
|
||||
# los gestiona Traefik/Coolify y no hay que publicar nada (§2.3 del contrato).
|
||||
# - monta ./init-db.sql por ruta local: imposible en Coolify, el compose se
|
||||
# guarda como docker_compose_raw en la DB, no hay árbol de archivos. No hace
|
||||
# falta: ensureDBExists() (pkg/config/config.go) crea la DB del DSN al arrancar.
|
||||
# - usa vars que el código de 0.7.2 ya no lee (WADEBUG/LOGTYPE); las reales son
|
||||
# DEBUG_ENABLED/LOG_TYPE (pkg/config/env/env.go).
|
||||
#
|
||||
# Detalles verificados contra el código 0.7.2:
|
||||
# - GateMiddleware (pkg/core/c0.go) devuelve 503 en TODO hasta activar licencia,
|
||||
# EXCEPTO /server/ok, /manager, /assets, /license/*, /swagger, /ws. El
|
||||
# healthcheck usa /server/ok (200 siempre) para no dejar al contenedor sin
|
||||
# ruta en Traefik antes de activar la licencia desde el Manager.
|
||||
# - whatsmeow guarda las sesiones SQLite en /app/dbdata (exPath = /app):
|
||||
# SIEMPRE volumen con nombre o cada redeploy desvincula los teléfonos.
|
||||
# - POSTGRES_AUTH_DB (URI completa) es OBLIGATORIA aunque parezca opcional:
|
||||
# con string vacío initPostgresAuthDB() devuelve (nil, nil) —sin error— y
|
||||
# NewPollService() hace panic por nil deref sobre el *sql.DB (bug upstream
|
||||
# 0.7.2, cmd/evolution-go/main.go:300 + pkg/poll/service/poll_service.go:39).
|
||||
# La URI interpola ${POSTGRES_PASSWORD}: Coolify sustituye al deployear,
|
||||
# el secreto vive solo en la env del servicio.
|
||||
# - POSTGRES_USERS_DB en URI para seguir el contrato upstream (con las vars
|
||||
# discretas también funciona; la DB evogo_users se auto-crea igual).
|
||||
# - DATABASE_SAVE_MESSAGES y GLOBAL_API_KEY son obligatorias (panicIfEmpty).
|
||||
# - SERVER_PORT no tiene default en el código: fijarlo siempre.
|
||||
#
|
||||
# Contrato: docs/AGENTS-coolify-apps.md
|
||||
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
|
||||
# - DB hermana por nombre de servicio, nunca localhost (§2.1)
|
||||
# - secretos (${VAR} sin default) llegan como env de Coolify, jamás aquí (§2.5)
|
||||
# - healthchecks con start_period holgado: el primer arranque aquí tarda
|
||||
# minutos (docs/casos/coolify-servicio-nuevo-503-no-available-server.md)
|
||||
|
||||
services:
|
||||
evolution-go:
|
||||
image: 'evoapicloud/evolution-go:0.7.2'
|
||||
environment:
|
||||
SERVER_PORT: '8080'
|
||||
CLIENT_NAME: '${CLIENT_NAME:-evolution}'
|
||||
# Obligatoria (panicIfEmpty). Secreto: inyectada por Coolify.
|
||||
GLOBAL_API_KEY: '${GLOBAL_API_KEY}'
|
||||
# Obligatoria y no vacía (panicIfEmpty).
|
||||
DATABASE_SAVE_MESSAGES: '${DATABASE_SAVE_MESSAGES:-false}'
|
||||
# OBLIGATORIA en URI (ver cabecera: vacía = panic nil deref en 0.7.2).
|
||||
# Coolify interpola ${POSTGRES_PASSWORD} del env del servicio al deploy.
|
||||
POSTGRES_AUTH_DB: 'postgresql://${POSTGRES_USER:-evolution}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_auth?sslmode=disable'
|
||||
POSTGRES_USERS_DB: 'postgresql://${POSTGRES_USER:-evolution}:${POSTGRES_PASSWORD}@evolution-postgres:5432/evogo_users?sslmode=disable'
|
||||
POSTGRES_HOST: evolution-postgres
|
||||
POSTGRES_PORT: '5432'
|
||||
POSTGRES_USER: '${POSTGRES_USER:-evolution}'
|
||||
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD}'
|
||||
POSTGRES_DB: '${POSTGRES_DB:-evogo_users}'
|
||||
# Nombres reales en 0.7.2 (env.go): DEBUG_ENABLED / LOG_TYPE.
|
||||
DEBUG_ENABLED: '${DEBUG_ENABLED:-INFO}'
|
||||
LOG_TYPE: '${LOG_TYPE:-console}'
|
||||
WEBHOOK_FILES: 'true'
|
||||
CONNECT_ON_STARTUP: 'false'
|
||||
OS_NAME: 'Linux'
|
||||
EVENT_IGNORE_GROUP: 'false'
|
||||
EVENT_IGNORE_STATUS: 'true'
|
||||
QRCODE_MAX_COUNT: '5'
|
||||
# AMQP/NATS/MinIO/WEBHOOK_URL apagados a propósito: nada de colas ni
|
||||
# objeto-storage extra en este host; el Manager funciona igual.
|
||||
AMQP_GLOBAL_ENABLED: 'false'
|
||||
NATS_GLOBAL_ENABLED: 'false'
|
||||
MINIO_ENABLED: 'false'
|
||||
# Sin bloque `ports:` — Traefik llega al 8080 interno (§2.3). FQDN sin
|
||||
# puerto + único `expose`: el mismo patrón verificado del stack firecrawl.
|
||||
expose:
|
||||
- '8080'
|
||||
volumes:
|
||||
# Sesiones whatsmeow (SQLite): perder esto = re-escanear QR de todos los
|
||||
# teléfonos en cada redeploy (§5).
|
||||
- 'evolution-data:/app/dbdata'
|
||||
- 'evolution-logs:/app/logs'
|
||||
depends_on:
|
||||
evolution-postgres:
|
||||
condition: service_healthy
|
||||
# Imagen alpine: wget es el de busybox (curl no viene instalado en 0.7.2).
|
||||
# /server/ok responde 200 sin licencia activa — ver comentario de cabecera.
|
||||
healthcheck:
|
||||
test: ['CMD', 'wget', '-q', '-O', '/dev/null', 'http://127.0.0.1:8080/server/ok']
|
||||
interval: 15s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 300s
|
||||
mem_limit: 1g
|
||||
memswap_limit: 1g
|
||||
cpus: 1.5
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 10m
|
||||
max-file: '3'
|
||||
|
||||
evolution-postgres:
|
||||
image: 'postgres:16-alpine'
|
||||
environment:
|
||||
POSTGRES_USER: '${POSTGRES_USER:-evolution}'
|
||||
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD}'
|
||||
POSTGRES_DB: '${POSTGRES_DB:-evogo_users}'
|
||||
expose:
|
||||
- '5432'
|
||||
healthcheck:
|
||||
test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-evolution} -d ${POSTGRES_DB:-evogo_users}']
|
||||
interval: 15s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 180s
|
||||
volumes:
|
||||
- 'evolution-postgres:/var/lib/postgresql/data'
|
||||
mem_limit: 1g
|
||||
memswap_limit: 1g
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 5m
|
||||
max-file: '2'
|
||||
|
||||
volumes:
|
||||
evolution-data: {}
|
||||
evolution-logs: {}
|
||||
evolution-postgres: {}
|
||||
@@ -0,0 +1,189 @@
|
||||
# Firecrawl - stack mínimo con imágenes precompiladas, para el host casero.
|
||||
#
|
||||
# Por qué no se usa el compose de upstream (firecrawl/firecrawl, rama main):
|
||||
# - construye 3 servicios desde fuente (apps/api, apps/playwright-service-ts,
|
||||
# apps/nuq-postgres). Compilar Chromium en un disco a ~26 ms/escritura es el
|
||||
# peor caso posible en este host.
|
||||
# - añade FoundationDB (+ un init) que solo se usan si NUQ_BACKEND está puesto.
|
||||
# - pide mem_limit 8G en api y 4G en playwright. El LXC tiene 4 cores.
|
||||
#
|
||||
# Aquí todo son imágenes ya publicadas: cero builds. FoundationDB queda fuera y
|
||||
# NUQ_BACKEND se deja vacío, que es su modo por defecto.
|
||||
#
|
||||
# Contrato: docs/AGENTS-coolify-apps.md
|
||||
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
|
||||
# - hermanos por nombre de servicio, nunca localhost (§2.1)
|
||||
# - volumen con nombre para lo que debe sobrevivir a un redeploy (§5)
|
||||
# - healthchecks con start_period holgado: el primer arranque aquí tarda
|
||||
# minutos (ver docs/casos/coolify-servicio-nuevo-503-no-available-server.md)
|
||||
# - sin secretos en el archivo: llegan como variables de entorno (§2.5)
|
||||
|
||||
services:
|
||||
api:
|
||||
image: 'ghcr.io/firecrawl/firecrawl:2.10.19'
|
||||
environment:
|
||||
HOST: 0.0.0.0
|
||||
PORT: '3002'
|
||||
INTERNAL_PORT: '3002'
|
||||
WORKER_PORT: '3005'
|
||||
EXTRACT_WORKER_PORT: '3004'
|
||||
ENV: local
|
||||
# Hermanos por nombre de servicio (§2.1)
|
||||
REDIS_URL: 'redis://redis:6379'
|
||||
REDIS_RATE_LIMIT_URL: 'redis://redis:6379'
|
||||
PLAYWRIGHT_MICROSERVICE_URL: 'http://playwright-service:3000/scrape'
|
||||
NUQ_RABBITMQ_URL: 'amqp://rabbitmq:5672'
|
||||
POSTGRES_HOST: nuq-postgres
|
||||
POSTGRES_PORT: '5432'
|
||||
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
|
||||
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
|
||||
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
|
||||
USE_DB_AUTHENTICATION: 'false'
|
||||
# Vacío a propósito: con NUQ_BACKEND sin definir, FoundationDB no se usa.
|
||||
NUQ_BACKEND: ''
|
||||
# Concurrencia recortada para 4 cores (upstream trae 8/10/5/5).
|
||||
NUM_WORKERS_PER_QUEUE: '${NUM_WORKERS_PER_QUEUE:-2}'
|
||||
CRAWL_CONCURRENT_REQUESTS: '${CRAWL_CONCURRENT_REQUESTS:-3}'
|
||||
MAX_CONCURRENT_JOBS: '${MAX_CONCURRENT_JOBS:-2}'
|
||||
BROWSER_POOL_SIZE: '${BROWSER_POOL_SIZE:-2}'
|
||||
HARNESS_STARTUP_TIMEOUT_MS: '${HARNESS_STARTUP_TIMEOUT_MS:-180000}'
|
||||
LOGGING_LEVEL: '${LOGGING_LEVEL:-info}'
|
||||
# Sin esto el worker responde "Can't accept connection due to RAM/CPU
|
||||
# load" y rechaza todo: el umbral por defecto (0.8) se supera constantemente
|
||||
# en un host compartido como este.
|
||||
MAX_RAM: '${MAX_RAM:-0.95}'
|
||||
MAX_CPU: '${MAX_CPU:-0.95}'
|
||||
# Secretos: inyectados por Coolify, nunca literales aquí (§2.5)
|
||||
BULL_AUTH_KEY: '${BULL_AUTH_KEY}'
|
||||
TEST_API_KEY: '${TEST_API_KEY}'
|
||||
OPENAI_API_KEY: '${OPENAI_API_KEY}'
|
||||
OPENAI_BASE_URL: '${OPENAI_BASE_URL}'
|
||||
MODEL_NAME: '${MODEL_NAME}'
|
||||
MODEL_EMBEDDING_NAME: '${MODEL_EMBEDDING_NAME}'
|
||||
SEARXNG_ENDPOINT: '${SEARXNG_ENDPOINT}'
|
||||
# Sin bloque `ports:` — Traefik llega al puerto interno (§2.3)
|
||||
expose:
|
||||
- '3002'
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
playwright-service:
|
||||
condition: service_started
|
||||
rabbitmq:
|
||||
condition: service_healthy
|
||||
nuq-postgres:
|
||||
condition: service_healthy
|
||||
# Verificado dentro de la imagen: NO trae wget ni nc, solo curl. Y /test,
|
||||
# /health y /v1/health dan 404; la raiz da 200. Un healthcheck con wget
|
||||
# falla siempre y deja el contenedor sin ruta en Traefik -> 503.
|
||||
healthcheck:
|
||||
test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/']
|
||||
interval: 15s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 300s
|
||||
ulimits:
|
||||
nofile:
|
||||
soft: 65535
|
||||
hard: 65535
|
||||
extra_hosts:
|
||||
- 'host.docker.internal:host-gateway'
|
||||
mem_limit: 3g
|
||||
memswap_limit: 3g
|
||||
cpus: 2.0
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 10m
|
||||
max-file: '3'
|
||||
|
||||
playwright-service:
|
||||
image: 'ghcr.io/firecrawl/playwright-service:latest'
|
||||
environment:
|
||||
PORT: '3000'
|
||||
MAX_CONCURRENT_PAGES: '${CRAWL_CONCURRENT_REQUESTS:-3}'
|
||||
BLOCK_MEDIA: '${BLOCK_MEDIA:-true}'
|
||||
ALLOW_LOCAL_WEBHOOKS: '${ALLOW_LOCAL_WEBHOOKS:-false}'
|
||||
expose:
|
||||
- '3000'
|
||||
# Chromium escribe mucho en /tmp; en tmpfs no toca el disco lento.
|
||||
tmpfs:
|
||||
- '/tmp/.cache:noexec,nosuid,size=512m'
|
||||
mem_limit: 2g
|
||||
memswap_limit: 2g
|
||||
cpus: 1.5
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 10m
|
||||
max-file: '3'
|
||||
|
||||
redis:
|
||||
image: 'redis:alpine'
|
||||
command: 'redis-server --bind 0.0.0.0 --save "" --appendonly no'
|
||||
expose:
|
||||
- '6379'
|
||||
healthcheck:
|
||||
test: ['CMD', 'redis-cli', 'ping']
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 10
|
||||
start_period: 60s
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 5m
|
||||
max-file: '2'
|
||||
|
||||
rabbitmq:
|
||||
image: 'rabbitmq:3-management'
|
||||
expose:
|
||||
- '5672'
|
||||
healthcheck:
|
||||
test: ['CMD', 'rabbitmq-diagnostics', '-q', 'check_running']
|
||||
interval: 15s
|
||||
timeout: 15s
|
||||
retries: 20
|
||||
start_period: 180s
|
||||
volumes:
|
||||
- 'firecrawl-rabbitmq:/var/lib/rabbitmq'
|
||||
mem_limit: 1g
|
||||
memswap_limit: 1g
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 5m
|
||||
max-file: '2'
|
||||
|
||||
nuq-postgres:
|
||||
image: 'ghcr.io/firecrawl/nuq-postgres:latest'
|
||||
environment:
|
||||
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
|
||||
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
|
||||
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
|
||||
expose:
|
||||
- '5432'
|
||||
healthcheck:
|
||||
test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-postgres} -d ${POSTGRES_DB:-postgres}']
|
||||
interval: 15s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 180s
|
||||
volumes:
|
||||
- 'firecrawl-postgres:/var/lib/postgresql/data'
|
||||
mem_limit: 1g
|
||||
memswap_limit: 1g
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: 10m
|
||||
max-file: '3'
|
||||
|
||||
volumes:
|
||||
firecrawl-postgres: {}
|
||||
firecrawl-rabbitmq: {}
|
||||
@@ -0,0 +1,6 @@
|
||||
.git
|
||||
node_modules
|
||||
.next
|
||||
*.md
|
||||
.gg
|
||||
tests
|
||||
@@ -0,0 +1,34 @@
|
||||
# oh-daddy (Next.js 16) - production image, built on the Coolify server.
|
||||
# No upstream Dockerfile exists (repo targets Railway/nixpacks), so this
|
||||
# replicates railway.json's startCommand contract: bash scripts/start.sh
|
||||
# (backgrounds the Inngest post-deploy re-sync, then execs `npm run start`).
|
||||
FROM node:22-alpine AS deps
|
||||
WORKDIR /app
|
||||
COPY package.json package-lock.json ./
|
||||
RUN npm ci
|
||||
|
||||
FROM node:22-alpine AS proddeps
|
||||
WORKDIR /app
|
||||
COPY package.json package-lock.json ./
|
||||
RUN npm ci --omit=dev
|
||||
|
||||
FROM node:22-alpine AS build
|
||||
WORKDIR /app
|
||||
ENV NEXT_TELEMETRY_DISABLED=1
|
||||
COPY --from=deps /app/node_modules ./node_modules
|
||||
COPY . .
|
||||
ARG NEXT_PUBLIC_APP_URL
|
||||
ENV NEXT_PUBLIC_APP_URL=$NEXT_PUBLIC_APP_URL
|
||||
RUN npm run build
|
||||
|
||||
FROM node:22-alpine AS runner
|
||||
WORKDIR /app
|
||||
RUN apk add --no-cache bash
|
||||
ENV NODE_ENV=production PORT=3000 NEXT_TELEMETRY_DISABLED=1
|
||||
COPY --from=proddeps /app/node_modules ./node_modules
|
||||
COPY --from=build /app/.next ./.next
|
||||
COPY --from=build /app/public ./public
|
||||
COPY --from=build /app/package.json ./package.json
|
||||
COPY --from=build /app/scripts ./scripts
|
||||
EXPOSE 3000
|
||||
CMD ["bash", "scripts/start.sh"]
|
||||
@@ -0,0 +1,108 @@
|
||||
# oh-daddy stack for Coolify (service type: docker compose, prebuilt/local images)
|
||||
# Upstream: https://github.com/KenKaiii/oh-daddy
|
||||
# App image `oh-daddy-app:local` is built on the server (Deploy-OhDaddy flow,
|
||||
# same pattern as Deploy-SoloLeveling.ps1) - Coolify's own deploy cannot pull it.
|
||||
# All secrets come from Coolify service env vars (is_literal) / .env in the
|
||||
# service dir; nothing is hardcoded here.
|
||||
#
|
||||
# Hard rules honored (docs/AGENTS-coolify-apps.md):
|
||||
# - no published 80/443; Traefik routes via SERVICE_FQDN_APP_3000 -> port 3000
|
||||
# - siblings reached by Docker service name (db, inngest, inngest-db, inngest-redis)
|
||||
# - named volumes for everything that must survive redeploys
|
||||
|
||||
services:
|
||||
app:
|
||||
image: oh-daddy-app:local
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- SERVICE_FQDN_APP_3000
|
||||
- PORT=3000
|
||||
- NODE_ENV=production
|
||||
- DATABASE_URL=${DATABASE_URL}
|
||||
- APP_ENCRYPTION_KEY=${APP_ENCRYPTION_KEY}
|
||||
- ADMIN_PASSWORD=${ADMIN_PASSWORD}
|
||||
- INNGEST_BASE_URL=${INNGEST_BASE_URL}
|
||||
- INNGEST_SIGNING_KEY=${INNGEST_SIGNING_KEY}
|
||||
- INNGEST_EVENT_KEY=${INNGEST_EVENT_KEY}
|
||||
- NEXT_PUBLIC_APP_URL=${NEXT_PUBLIC_APP_URL}
|
||||
expose:
|
||||
- "3000"
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-qO-", "http://127.0.0.1:3000/login"]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 60s
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
|
||||
db:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- POSTGRES_DB=ohdaddy
|
||||
- POSTGRES_USER=ohdaddy
|
||||
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD}
|
||||
volumes:
|
||||
- oh-daddy-db-data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U ohdaddy -d ohdaddy"]
|
||||
interval: 5s
|
||||
timeout: 5s
|
||||
retries: 12
|
||||
|
||||
# Self-hosted Inngest engine (NOT Inngest Cloud) - internal only, no FQDN.
|
||||
inngest:
|
||||
image: inngest/inngest:v1.44.0
|
||||
restart: unless-stopped
|
||||
command: ["inngest", "start"]
|
||||
environment:
|
||||
- INNGEST_SIGNING_KEY=${INNGEST_SIGNING_KEY}
|
||||
- INNGEST_EVENT_KEY=${INNGEST_EVENT_KEY}
|
||||
- INNGEST_POSTGRES_URI=${INNGEST_POSTGRES_URI}
|
||||
- INNGEST_REDIS_URI=redis://inngest-redis:6379
|
||||
expose:
|
||||
- "8288"
|
||||
healthcheck:
|
||||
test: ["CMD", "inngest", "alpha", "doctor", "healthcheck"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 40s
|
||||
depends_on:
|
||||
inngest-db:
|
||||
condition: service_healthy
|
||||
inngest-redis:
|
||||
condition: service_healthy
|
||||
|
||||
inngest-db:
|
||||
image: postgres:17-alpine
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- POSTGRES_DB=inngest
|
||||
- POSTGRES_USER=inngest
|
||||
- POSTGRES_PASSWORD=${INNGEST_DB_PASSWORD}
|
||||
volumes:
|
||||
- oh-daddy-inngest-pg-data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U inngest -d inngest"]
|
||||
interval: 5s
|
||||
timeout: 5s
|
||||
retries: 12
|
||||
|
||||
inngest-redis:
|
||||
image: redis:7-alpine
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- oh-daddy-inngest-redis-data:/data
|
||||
healthcheck:
|
||||
test: ["CMD", "redis-cli", "ping"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
retries: 5
|
||||
|
||||
volumes:
|
||||
oh-daddy-db-data: {}
|
||||
oh-daddy-inngest-pg-data: {}
|
||||
oh-daddy-inngest-redis-data: {}
|
||||
@@ -0,0 +1,22 @@
|
||||
# OpenSEO — env vars para Coolify (template, sin secretos reales).
|
||||
#
|
||||
# Copia este archivo a `stacks/open-seo/.env.coolify`, rellena lo que aplique,
|
||||
# y pásalo a `New-CoolifyService.ps1 -EnvFile`. El script empuja cada línea
|
||||
# KEY=VALUE como env var del servicio; las referencias `${VAR}` dentro del
|
||||
# compose se resuelven en runtime desde esas vars.
|
||||
#
|
||||
# Nunca commitees el archivo `.env.coolify` real — está en .gitignore.
|
||||
|
||||
# SEO data (opcional). Base64 de `email:password` de dataforseo.com — NO la
|
||||
# dashboard API key. Vacío = la app arranca, los workflows SEO muestran "no
|
||||
# data". Ver https://openseo.so/docs/DATAFORSEO_API_KEY.md
|
||||
DATAFORSEO_API_KEY=
|
||||
|
||||
# Telemetry opt-out. "1" para desactivar el heartbeat anónimo y el beacon de
|
||||
# fallo de preflight. Por defecto encendido.
|
||||
OPENSEO_TELEMETRY_DISABLED=1
|
||||
DO_NOT_TRACK=1
|
||||
|
||||
# SAM, el agente SEO dentro de la app (opcional). Oculto si vacío.
|
||||
OPENROUTER_API_KEY=
|
||||
OPENROUTER_MODEL=
|
||||
@@ -0,0 +1,154 @@
|
||||
# OpenSEO — stack self-host en Coolify
|
||||
|
||||
> Resuelto el **2026-08-27** contra el host real.
|
||||
> Target: **LXC 102** (`coolify`), servicio compose, FQDN
|
||||
> **`https://openseo.urieljareth.org`**.
|
||||
> Imagen: **`ghcr.io/every-app/open-seo:sha-c469a48`** (mismo SHA que el deploy
|
||||
> fallido anterior — esta vez la build de Vite sí corre, en el entrypoint).
|
||||
|
||||
---
|
||||
|
||||
## Por qué existe este stack
|
||||
|
||||
El deploy previo (`uuid kj0kccsb4d46tm0d6qe6docy`, `name=open-seo:main-...`,
|
||||
deployment `fkojyfkqzp69hcba6miy8oer`) terminó con Coolify marcando verde y
|
||||
**Caddy respondiendo 404 a todo**. Diagnóstico:
|
||||
|
||||
- `build_pack=railpack` clonó `every-app/open-seo@main`, construyó imagen local
|
||||
con el mismo SHA `c469a48ae90ab58413b198fe3d1ac1aa90a9b070` y la cacheó.
|
||||
- En el redeploy: `No build configuration changed & image found (...) Build
|
||||
step skipped` → la imagen cacheada **no tenía `/app/dist`** (los artefactos
|
||||
del build de Vite) y Coolify la reusó.
|
||||
- Caddy (`/Caddyfile` con `root * /app/dist` + SPA fallback a `/index.html`)
|
||||
no encontró nada y devolvió 404 a `/`, `/robots.txt`, `/health`.
|
||||
- El README upstream lo dice textual: *"We recommend self-hosting with
|
||||
Cloudflare as opposed to Railway, Coolify or Dokploy. We plan to make it
|
||||
simpler to host on those platforms in the next few months."*
|
||||
|
||||
**Solución:** dejar de seguir upstream y consumir la **imagen prebuilt** que el
|
||||
propio equipo publica en GHCR. Esa imagen tiene la cadena correcta:
|
||||
`docker-entrypoint.sh` corre preflight → migrations → `pnpm run build` (que sí
|
||||
genera `/app/dist`) → `vite preview` en el puerto `3001`. Y usa un fingerprint
|
||||
para no reconstruir cuando los env vars relevantes no cambiaron.
|
||||
|
||||
Stack: **un solo servicio** (OpenSEO es self-contained: SQLite vía workerd en
|
||||
`/app/.wrangler`, volumen `openseo-data`). Sin DB externa.
|
||||
|
||||
---
|
||||
|
||||
## Archivos
|
||||
|
||||
| Archivo | Para qué |
|
||||
|---|---|
|
||||
| `docker-compose.coolify.yml` | Compose que consume Coolify vía `POST /services` |
|
||||
| `.env.example` | Template de env vars (sin secretos) |
|
||||
| `.env.coolify` | **No committed.** Lo crea el operador con `cp .env.example .env.coolify` y rellena |
|
||||
|
||||
---
|
||||
|
||||
## Variables de entorno
|
||||
|
||||
Hardcoded en el compose (porque son decisión de arquitectura, no secretos):
|
||||
|
||||
| Var | Valor | Por qué |
|
||||
|---|---|---|
|
||||
| `PORT` | `3001` | Es donde escucha `vite preview` (per `Dockerfile.selfhost`) |
|
||||
| `AUTH_MODE` | `local_noauth` | Single admin, sin pantalla de login. Aquí no tenemos `TEAM_DOMAIN`/`POLICY_AUD` de Cloudflare Access |
|
||||
| `ALLOWED_HOST` | `openseo.urieljareth.org` | Sin esto, Vite bloquea toda petición externa con "Blocked request" |
|
||||
| `CLOUDFLARE_INCLUDE_PROCESS_ENV` | `true` | Lo exige el runtime workerd para que process.env llegue a los bindings |
|
||||
|
||||
Suministradas vía `.env.coolify` (env vars del servicio en Coolify):
|
||||
|
||||
| Var | Default | Efecto |
|
||||
|---|---|---|
|
||||
| `DATAFORSEO_API_KEY` | vacío | **WARN** del preflight (no FAIL). Vacío = la app arranca, los workflows SEO devuelven "no data" |
|
||||
| `OPENSEO_TELEMETRY_DISABLED` | `1` | Apaga el heartbeat anónimo |
|
||||
| `DO_NOT_TRACK` | `1` | Alias del anterior |
|
||||
| `OPENROUTER_API_KEY` | vacío | Habilita a SAM (el agente SEO integrado) si se setea |
|
||||
| `OPENROUTER_MODEL` | vacío | Modelo a usar con SAM |
|
||||
|
||||
---
|
||||
|
||||
## Deploy
|
||||
|
||||
### 1. (Manual, una sola vez) Ingress del túnel de Cloudflare
|
||||
|
||||
El token de Cloudflare **no está** en `.env.local.ps1`, así que esto se hace en
|
||||
el dashboard:
|
||||
|
||||
1. Cloudflare Zero Trust → Networks → Tunnels → tunnel `urieljareth` →
|
||||
Configure → Public hostname.
|
||||
2. Add a public hostname:
|
||||
- Subdomain: `openseo`
|
||||
- Domain: `urieljareth.org`
|
||||
- Service: **HTTP** (no HTTPS, lo gestiona Coolify/Traefik)
|
||||
- URL: `coolify.urieljareth.org` (o la IP interna del proxy de Coolify —
|
||||
misma que usan los demás subdominios)
|
||||
|
||||
### 2. Crear el servicio en Coolify
|
||||
|
||||
```powershell
|
||||
. .\.env.local.ps1
|
||||
|
||||
# Crear el archivo de env real (gitignored)
|
||||
Copy-Item .\stacks\open-seo\.env.example .\stacks\open-seo\.env.coolify
|
||||
# Editar .\stacks\open-seo\.env.coolify si quieres setear DATAFORSEO_API_KEY
|
||||
|
||||
.\deploy_skill\scripts\New-CoolifyService.ps1 `
|
||||
-AppPath .\stacks\open-seo `
|
||||
-AppName open-seo `
|
||||
-Fqdn https://openseo.urieljareth.org `
|
||||
-PrimaryService app `
|
||||
-ProjectName "AI AGENCY" -EnvironmentName production `
|
||||
-EnvFile .\stacks\open-seo\.env.coolify `
|
||||
-InstantDeploy
|
||||
```
|
||||
|
||||
### 3. Esperar al primer arranque
|
||||
|
||||
El primer `up` tarda **1-2 min**: preflight + migrations + vite build + arranque
|
||||
de `vite preview`. Traefik no enruta hasta que el contenedor esté `healthy`.
|
||||
|
||||
```powershell
|
||||
.\coolify_skill\scripts\Test-CoolifyServiceReady.ps1 -Uuid <uuid> -WaitSeconds 600
|
||||
```
|
||||
|
||||
### 4. Verificar
|
||||
|
||||
```powershell
|
||||
# 1. Endpoint público responde (TLS emitido, Traefik enrutando)
|
||||
curl.exe -k -sSI https://openseo.urieljareth.org/
|
||||
|
||||
# 2. Status del contenedor
|
||||
.\coolify_skill\scripts\Get-CoolifyDockerStatus.ps1 -Filter openseo
|
||||
|
||||
# 3. Logs del entrypoint (debería verse "Preflight passed")
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker logs <cont> --tail 60"
|
||||
|
||||
# 4. Preflight reporta lo que falta
|
||||
.\scripts\Invoke-ProxmoxSsh.ps1 -Command "pct exec 102 -- docker exec <cont> wget -qO- http://127.0.0.1:3001/api/health"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Rollback / limpieza
|
||||
|
||||
| Acción | Comando |
|
||||
|---|---|
|
||||
| Parar la app rota original | `docker stop kj0kccsb4d46tm0d6qe6docy-055629993145` (vía `pct exec 102 --`) |
|
||||
| Borrar la app rota de Coolify | UI → Service `kj0kccsb4d46tm0d6qe6docy` → Delete |
|
||||
| Re-deployar | UI → Service `open-seo` → Redeploy |
|
||||
|
||||
---
|
||||
|
||||
## Cosas que caducan
|
||||
|
||||
- **El tag `:sha-c469a48` se queda viejo.** Cuando el upstream publique un SHA
|
||||
más reciente, actualizar el `image:` en `docker-compose.coolify.yml` y
|
||||
redeployar. `v0.1.6` también existe (publicado 8 días antes).
|
||||
- **El entrypoint vuelve a buildear `dist`** cada vez que algún env var del
|
||||
prefijo `VITE_*` / `AUTH_MODE` / `POSTHOG_*` / `TURNSTILE_SITE_KEY` /
|
||||
`BYPASS_EMAIL_VERIFICATION` cambie. Es intencional — el fingerprint está
|
||||
ahí para no rehacer cuando nada relevante cambió.
|
||||
- **Coolify normaliza el compose al guardarlo y borra los comentarios.** La
|
||||
versión con explicaciones es la del repo, no la que se ve en la UI.
|
||||
@@ -0,0 +1,78 @@
|
||||
# OpenSEO self-host — Coolify stack (LXC 102)
|
||||
#
|
||||
# Por qué este compose en lugar del repo upstream (`every-app/open-seo`) vía
|
||||
# railpack:
|
||||
# - el deploy anterior (uuid kj0kccsb4d46tm0d6qe6docy) terminó con Caddy
|
||||
# respondiendo 404 a todo porque /app/dist no existía en la imagen cacheada
|
||||
# (Coolify saltó el build: "No build configuration changed & image found ...
|
||||
# Build step skipped").
|
||||
# - el README upstream lo dice explícitamente: "We recommend self-hosting
|
||||
# with Cloudflare as opposed to Railway, Coolify or Dokploy. We plan to
|
||||
# make it simpler to host on those platforms in the next few months."
|
||||
# - esta imagen (`ghcr.io/every-app/open-seo:sha-c469a48`) es la build del
|
||||
# mismo commit, pero el build de Vite corre en `docker-entrypoint.sh` al
|
||||
# arrancar el contenedor, no en el build de la imagen. Y el entrypoint ya
|
||||
# tiene la lógica de fingerprint para no reconstruir cuando los env vars
|
||||
# relevantes no cambiaron.
|
||||
#
|
||||
# Contrato: docs/AGENTS-coolify-apps.md
|
||||
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
|
||||
# - SECRETOS vía env vars inyectados por Coolify (§2.5)
|
||||
# - volumen con nombre para /app/.wrangler (SQLite que sobrevive al redeploy, §5)
|
||||
# - healthcheck independiente de servicios externos al boot
|
||||
# - AUTH_MODE=local_noauth porque aquí no hay TEAM_DOMAIN/POLICY_AUD de
|
||||
# Cloudflare Access. Si más adelante se quiere proteger con auth, cambiar
|
||||
# a AUTH_MODE=cloudflare_access + TEAM_DOMAIN + POLICY_AUD.
|
||||
# - ALLOWED_HOST es obligatorio detrás del túnel: sin él Vite bloquea toda
|
||||
# petición externa con "Blocked request" (ver preflight info level).
|
||||
|
||||
services:
|
||||
app:
|
||||
image: 'ghcr.io/every-app/open-seo:sha-c469a48'
|
||||
environment:
|
||||
# Puerto en el que escucha `vite preview` (per Dockerfile.selfhost / entrypoint).
|
||||
- PORT=3001
|
||||
# Single admin user, sin pantalla de login. NO exponer públicamente sin
|
||||
# poner tu propia auth delante — el preflight lo dice literal.
|
||||
- AUTH_MODE=local_noauth
|
||||
# Host header permitido. Es el FQDN público por el que llega el tráfico
|
||||
# desde el túnel de Cloudflare.
|
||||
- ALLOWED_HOST=openseo.urieljareth.org
|
||||
# Requerido por el runtime workerd para exponer process.env a los
|
||||
# bindings (lo exige el compose upstream).
|
||||
- CLOUDFLARE_INCLUDE_PROCESS_ENV=true
|
||||
# SEO data (opcional). Vacío = la app arranca, los workflows SEO
|
||||
# devuelven "no data". Se setea después vía Coolify env.
|
||||
- DATAFORSEO_API_KEY=${DATAFORSEO_API_KEY:-}
|
||||
# Telemetry opt-out (también vía DO_NOT_TRACK). Por defecto apagado.
|
||||
- OPENSEO_TELEMETRY_DISABLED=${OPENSEO_TELEMETRY_DISABLED:-}
|
||||
- DO_NOT_TRACK=${DO_NOT_TRACK:-}
|
||||
# AI features (SAM, el agente SEO integrado). Vacío = SAM deshabilitado.
|
||||
- OPENROUTER_API_KEY=${OPENROUTER_API_KEY:-}
|
||||
- OPENROUTER_MODEL=${OPENROUTER_MODEL:-}
|
||||
expose:
|
||||
- '3001'
|
||||
# El endpoint /api/health lo sirve el propio preflight (ver
|
||||
# src/lib/selfhost-preflight.ts) sin auth — seguro para el healthcheck.
|
||||
# Usamos `node` directamente porque la imagen es node:22 y el HEALTHCHECK
|
||||
# upstream hace exactamente esto.
|
||||
healthcheck:
|
||||
test:
|
||||
- CMD-SHELL
|
||||
- "node -e \"fetch('http://127.0.0.1:3001/api/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))\""
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
# Primer arranque: preflight + migrations + vite build (1-2 min).
|
||||
start_period: 300s
|
||||
volumes:
|
||||
- 'openseo-data:/app/.wrangler'
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: '10m'
|
||||
max-file: '3'
|
||||
|
||||
volumes:
|
||||
openseo-data: {}
|
||||
Reference in New Issue
Block a user