Actualiza toolkit operativo y documentación

This commit is contained in:
urieljareth
2026-09-10 20:53:50 -06:00
parent 3b7209dcc1
commit 714057bfc8
69 changed files with 6023 additions and 384 deletions
+189
View File
@@ -0,0 +1,189 @@
# Firecrawl - stack mínimo con imágenes precompiladas, para el host casero.
#
# Por qué no se usa el compose de upstream (firecrawl/firecrawl, rama main):
# - construye 3 servicios desde fuente (apps/api, apps/playwright-service-ts,
# apps/nuq-postgres). Compilar Chromium en un disco a ~26 ms/escritura es el
# peor caso posible en este host.
# - añade FoundationDB (+ un init) que solo se usan si NUQ_BACKEND está puesto.
# - pide mem_limit 8G en api y 4G en playwright. El LXC tiene 4 cores.
#
# Aquí todo son imágenes ya publicadas: cero builds. FoundationDB queda fuera y
# NUQ_BACKEND se deja vacío, que es su modo por defecto.
#
# Contrato: docs/AGENTS-coolify-apps.md
# - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3)
# - hermanos por nombre de servicio, nunca localhost (§2.1)
# - volumen con nombre para lo que debe sobrevivir a un redeploy (§5)
# - healthchecks con start_period holgado: el primer arranque aquí tarda
# minutos (ver docs/casos/coolify-servicio-nuevo-503-no-available-server.md)
# - sin secretos en el archivo: llegan como variables de entorno (§2.5)
services:
api:
image: 'ghcr.io/firecrawl/firecrawl:2.10.19'
environment:
HOST: 0.0.0.0
PORT: '3002'
INTERNAL_PORT: '3002'
WORKER_PORT: '3005'
EXTRACT_WORKER_PORT: '3004'
ENV: local
# Hermanos por nombre de servicio (§2.1)
REDIS_URL: 'redis://redis:6379'
REDIS_RATE_LIMIT_URL: 'redis://redis:6379'
PLAYWRIGHT_MICROSERVICE_URL: 'http://playwright-service:3000/scrape'
NUQ_RABBITMQ_URL: 'amqp://rabbitmq:5672'
POSTGRES_HOST: nuq-postgres
POSTGRES_PORT: '5432'
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
USE_DB_AUTHENTICATION: 'false'
# Vacío a propósito: con NUQ_BACKEND sin definir, FoundationDB no se usa.
NUQ_BACKEND: ''
# Concurrencia recortada para 4 cores (upstream trae 8/10/5/5).
NUM_WORKERS_PER_QUEUE: '${NUM_WORKERS_PER_QUEUE:-2}'
CRAWL_CONCURRENT_REQUESTS: '${CRAWL_CONCURRENT_REQUESTS:-3}'
MAX_CONCURRENT_JOBS: '${MAX_CONCURRENT_JOBS:-2}'
BROWSER_POOL_SIZE: '${BROWSER_POOL_SIZE:-2}'
HARNESS_STARTUP_TIMEOUT_MS: '${HARNESS_STARTUP_TIMEOUT_MS:-180000}'
LOGGING_LEVEL: '${LOGGING_LEVEL:-info}'
# Sin esto el worker responde "Can't accept connection due to RAM/CPU
# load" y rechaza todo: el umbral por defecto (0.8) se supera constantemente
# en un host compartido como este.
MAX_RAM: '${MAX_RAM:-0.95}'
MAX_CPU: '${MAX_CPU:-0.95}'
# Secretos: inyectados por Coolify, nunca literales aquí (§2.5)
BULL_AUTH_KEY: '${BULL_AUTH_KEY}'
TEST_API_KEY: '${TEST_API_KEY}'
OPENAI_API_KEY: '${OPENAI_API_KEY}'
OPENAI_BASE_URL: '${OPENAI_BASE_URL}'
MODEL_NAME: '${MODEL_NAME}'
MODEL_EMBEDDING_NAME: '${MODEL_EMBEDDING_NAME}'
SEARXNG_ENDPOINT: '${SEARXNG_ENDPOINT}'
# Sin bloque `ports:` — Traefik llega al puerto interno (§2.3)
expose:
- '3002'
depends_on:
redis:
condition: service_started
playwright-service:
condition: service_started
rabbitmq:
condition: service_healthy
nuq-postgres:
condition: service_healthy
# Verificado dentro de la imagen: NO trae wget ni nc, solo curl. Y /test,
# /health y /v1/health dan 404; la raiz da 200. Un healthcheck con wget
# falla siempre y deja el contenedor sin ruta en Traefik -> 503.
healthcheck:
test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/']
interval: 15s
timeout: 10s
retries: 20
start_period: 300s
ulimits:
nofile:
soft: 65535
hard: 65535
extra_hosts:
- 'host.docker.internal:host-gateway'
mem_limit: 3g
memswap_limit: 3g
cpus: 2.0
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
playwright-service:
image: 'ghcr.io/firecrawl/playwright-service:latest'
environment:
PORT: '3000'
MAX_CONCURRENT_PAGES: '${CRAWL_CONCURRENT_REQUESTS:-3}'
BLOCK_MEDIA: '${BLOCK_MEDIA:-true}'
ALLOW_LOCAL_WEBHOOKS: '${ALLOW_LOCAL_WEBHOOKS:-false}'
expose:
- '3000'
# Chromium escribe mucho en /tmp; en tmpfs no toca el disco lento.
tmpfs:
- '/tmp/.cache:noexec,nosuid,size=512m'
mem_limit: 2g
memswap_limit: 2g
cpus: 1.5
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
redis:
image: 'redis:alpine'
command: 'redis-server --bind 0.0.0.0 --save "" --appendonly no'
expose:
- '6379'
healthcheck:
test: ['CMD', 'redis-cli', 'ping']
interval: 15s
timeout: 5s
retries: 10
start_period: 60s
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 5m
max-file: '2'
rabbitmq:
image: 'rabbitmq:3-management'
expose:
- '5672'
healthcheck:
test: ['CMD', 'rabbitmq-diagnostics', '-q', 'check_running']
interval: 15s
timeout: 15s
retries: 20
start_period: 180s
volumes:
- 'firecrawl-rabbitmq:/var/lib/rabbitmq'
mem_limit: 1g
memswap_limit: 1g
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 5m
max-file: '2'
nuq-postgres:
image: 'ghcr.io/firecrawl/nuq-postgres:latest'
environment:
POSTGRES_USER: '${POSTGRES_USER:-postgres}'
POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}'
POSTGRES_DB: '${POSTGRES_DB:-postgres}'
expose:
- '5432'
healthcheck:
test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-postgres} -d ${POSTGRES_DB:-postgres}']
interval: 15s
timeout: 10s
retries: 20
start_period: 180s
volumes:
- 'firecrawl-postgres:/var/lib/postgresql/data'
mem_limit: 1g
memswap_limit: 1g
restart: unless-stopped
logging:
driver: json-file
options:
max-size: 10m
max-file: '3'
volumes:
firecrawl-postgres: {}
firecrawl-rabbitmq: {}