# Firecrawl - stack mínimo con imágenes precompiladas, para el host casero. # # Por qué no se usa el compose de upstream (firecrawl/firecrawl, rama main): # - construye 3 servicios desde fuente (apps/api, apps/playwright-service-ts, # apps/nuq-postgres). Compilar Chromium en un disco a ~26 ms/escritura es el # peor caso posible en este host. # - añade FoundationDB (+ un init) que solo se usan si NUQ_BACKEND está puesto. # - pide mem_limit 8G en api y 4G en playwright. El LXC tiene 4 cores. # # Aquí todo son imágenes ya publicadas: cero builds. FoundationDB queda fuera y # NUQ_BACKEND se deja vacío, que es su modo por defecto. # # Contrato: docs/AGENTS-coolify-apps.md # - sin publicar 80/443: solo `expose`, Traefik enruta (§2.3) # - hermanos por nombre de servicio, nunca localhost (§2.1) # - volumen con nombre para lo que debe sobrevivir a un redeploy (§5) # - healthchecks con start_period holgado: el primer arranque aquí tarda # minutos (ver docs/casos/coolify-servicio-nuevo-503-no-available-server.md) # - sin secretos en el archivo: llegan como variables de entorno (§2.5) services: api: image: 'ghcr.io/firecrawl/firecrawl:2.10.19' environment: HOST: 0.0.0.0 PORT: '3002' INTERNAL_PORT: '3002' WORKER_PORT: '3005' EXTRACT_WORKER_PORT: '3004' ENV: local # Hermanos por nombre de servicio (§2.1) REDIS_URL: 'redis://redis:6379' REDIS_RATE_LIMIT_URL: 'redis://redis:6379' PLAYWRIGHT_MICROSERVICE_URL: 'http://playwright-service:3000/scrape' NUQ_RABBITMQ_URL: 'amqp://rabbitmq:5672' POSTGRES_HOST: nuq-postgres POSTGRES_PORT: '5432' POSTGRES_USER: '${POSTGRES_USER:-postgres}' POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}' POSTGRES_DB: '${POSTGRES_DB:-postgres}' USE_DB_AUTHENTICATION: 'false' # Vacío a propósito: con NUQ_BACKEND sin definir, FoundationDB no se usa. NUQ_BACKEND: '' # Concurrencia recortada para 4 cores (upstream trae 8/10/5/5). NUM_WORKERS_PER_QUEUE: '${NUM_WORKERS_PER_QUEUE:-2}' CRAWL_CONCURRENT_REQUESTS: '${CRAWL_CONCURRENT_REQUESTS:-3}' MAX_CONCURRENT_JOBS: '${MAX_CONCURRENT_JOBS:-2}' BROWSER_POOL_SIZE: '${BROWSER_POOL_SIZE:-2}' HARNESS_STARTUP_TIMEOUT_MS: '${HARNESS_STARTUP_TIMEOUT_MS:-180000}' LOGGING_LEVEL: '${LOGGING_LEVEL:-info}' # Sin esto el worker responde "Can't accept connection due to RAM/CPU # load" y rechaza todo: el umbral por defecto (0.8) se supera constantemente # en un host compartido como este. MAX_RAM: '${MAX_RAM:-0.95}' MAX_CPU: '${MAX_CPU:-0.95}' # Secretos: inyectados por Coolify, nunca literales aquí (§2.5) BULL_AUTH_KEY: '${BULL_AUTH_KEY}' TEST_API_KEY: '${TEST_API_KEY}' OPENAI_API_KEY: '${OPENAI_API_KEY}' OPENAI_BASE_URL: '${OPENAI_BASE_URL}' MODEL_NAME: '${MODEL_NAME}' MODEL_EMBEDDING_NAME: '${MODEL_EMBEDDING_NAME}' SEARXNG_ENDPOINT: '${SEARXNG_ENDPOINT}' # Sin bloque `ports:` — Traefik llega al puerto interno (§2.3) expose: - '3002' depends_on: redis: condition: service_started playwright-service: condition: service_started rabbitmq: condition: service_healthy nuq-postgres: condition: service_healthy # Verificado dentro de la imagen: NO trae wget ni nc, solo curl. Y /test, # /health y /v1/health dan 404; la raiz da 200. Un healthcheck con wget # falla siempre y deja el contenedor sin ruta en Traefik -> 503. healthcheck: test: ['CMD', 'curl', '-fsS', '-o', '/dev/null', 'http://127.0.0.1:3002/'] interval: 15s timeout: 10s retries: 20 start_period: 300s ulimits: nofile: soft: 65535 hard: 65535 extra_hosts: - 'host.docker.internal:host-gateway' mem_limit: 3g memswap_limit: 3g cpus: 2.0 restart: unless-stopped logging: driver: json-file options: max-size: 10m max-file: '3' playwright-service: image: 'ghcr.io/firecrawl/playwright-service:latest' environment: PORT: '3000' MAX_CONCURRENT_PAGES: '${CRAWL_CONCURRENT_REQUESTS:-3}' BLOCK_MEDIA: '${BLOCK_MEDIA:-true}' ALLOW_LOCAL_WEBHOOKS: '${ALLOW_LOCAL_WEBHOOKS:-false}' expose: - '3000' # Chromium escribe mucho en /tmp; en tmpfs no toca el disco lento. tmpfs: - '/tmp/.cache:noexec,nosuid,size=512m' mem_limit: 2g memswap_limit: 2g cpus: 1.5 restart: unless-stopped logging: driver: json-file options: max-size: 10m max-file: '3' redis: image: 'redis:alpine' command: 'redis-server --bind 0.0.0.0 --save "" --appendonly no' expose: - '6379' healthcheck: test: ['CMD', 'redis-cli', 'ping'] interval: 15s timeout: 5s retries: 10 start_period: 60s restart: unless-stopped logging: driver: json-file options: max-size: 5m max-file: '2' rabbitmq: image: 'rabbitmq:3-management' expose: - '5672' healthcheck: test: ['CMD', 'rabbitmq-diagnostics', '-q', 'check_running'] interval: 15s timeout: 15s retries: 20 start_period: 180s volumes: - 'firecrawl-rabbitmq:/var/lib/rabbitmq' mem_limit: 1g memswap_limit: 1g restart: unless-stopped logging: driver: json-file options: max-size: 5m max-file: '2' nuq-postgres: image: 'ghcr.io/firecrawl/nuq-postgres:latest' environment: POSTGRES_USER: '${POSTGRES_USER:-postgres}' POSTGRES_PASSWORD: '${POSTGRES_PASSWORD:-postgres}' POSTGRES_DB: '${POSTGRES_DB:-postgres}' expose: - '5432' healthcheck: test: ['CMD-SHELL', 'pg_isready -U ${POSTGRES_USER:-postgres} -d ${POSTGRES_DB:-postgres}'] interval: 15s timeout: 10s retries: 20 start_period: 180s volumes: - 'firecrawl-postgres:/var/lib/postgresql/data' mem_limit: 1g memswap_limit: 1g restart: unless-stopped logging: driver: json-file options: max-size: 10m max-file: '3' volumes: firecrawl-postgres: {} firecrawl-rabbitmq: {}