# ============================================================================= # Firecrawl — self-hosted web scraper for Hermes (scrape/extract) + vLLM (AI) # Runs entirely on a private `firecrawl-net` bridge; only the API port # (FIRECRAWL_PORT=3002) is published to the host. The homepage's host :3000 is # untouched: playwright binds 3000 INTERNALLY only (firecrawl-net), and the # api's in-container workers bind the api container's own namespace — so # nothing firecrawl runs can collide with host :3000. No docker volumes: all # state lives under ./firecrawl-data/. Images are pulled (never # built) — see the FIRECRAWL_*_IMAGE digest pins in .env. # # Port layout: # host :3002 -> api container :3002 (the ONLY published port) # container-internal (firecrawl-net): # api (express) :3002 # playwright-service :3000 # extract-worker :3004 # queue-worker (liveness) :3005 # nuq-worker x5 :3006-3010 # nuq-prefetch-worker :3011 # nuq-reconciler-worker :3012 # cclog-worker :3013 # postgres :5432 / rabbitmq :5672 / redis :6380 — internal, not published # # Env: each firecrawl service loads ONLY its own env file — NEVER the global # stack .env (it would leak spotizerr's REDIS_PASSWORD into firecrawl-redis, # whose entrypoint turns it into requirepass and break the passwordless # REDIS_URL). .firecrawl.env carries the api+postgres contract; the two keys # needing ${} interpolation (OPENAI_BASE_URL -> vLLM on the LAN, # SEARXNG_ENDPOINT -> SearXNG on the LAN) live in `environment:` because # env_file does not interpolate. LAN reach: extra_hosts host-gateway. # # Key ordering per service (matches compose.yml): # image → container_name → env_file → networks/network_mode # → depends_on → cap_add → ports → volumes → environment # → labels → healthcheck → security_opt → mem_limit → cpus # → devices → restart # ============================================================================= networks: firecrawl-net: driver: bridge services: # --------------------------------------------------------------------------- # API + in-container workers (api, queue-worker, extract-worker, nuq workers) # --------------------------------------------------------------------------- firecrawl-api: image: ${FIRECRAWL_IMAGE} container_name: firecrawl-api env_file: - ./env/.firecrawl.env networks: - firecrawl-net depends_on: firecrawl-postgres: condition: service_healthy firecrawl-rabbitmq: condition: service_healthy firecrawl-redis: condition: service_healthy firecrawl-playwright: condition: service_healthy ulimits: nofile: soft: 65535 hard: 65535 extra_hosts: - "host.docker.internal:host-gateway" ports: - "${FIRECRAWL_PORT}:3002" # ${} expansion (env_file does not interpolate): vLLM is host-networked, so # LAN IP + VLLM_PORT reach it from inside; same for SearXNG. environment: - "OPENAI_BASE_URL=http://${LOCAL_IPV4}:${VLLM_PORT}/v1" - "SEARXNG_ENDPOINT=http://${LOCAL_IPV4}:${SEARXNG_PORT}" labels: - "autoheal=true" healthcheck: test: ["CMD", "curl", "-sf", "http://127.0.0.1:3002/v0/health/liveness"] interval: 15s timeout: 5s retries: 5 start_period: 60s logging: driver: json-file options: max-size: "10m" max-file: "3" compress: "true" security_opt: - no-new-privileges:true mem_limit: 6g memswap_limit: 8g cpus: 2.0 restart: unless-stopped # --------------------------------------------------------------------------- # Playwright browser microservice — internal only (port 3000 on firecrawl-net) # --------------------------------------------------------------------------- firecrawl-playwright: image: ${FIRECRAWL_PLAYWRIGHT_IMAGE} container_name: firecrawl-playwright env_file: - ./env/.firecrawl-playwright.env networks: - firecrawl-net cap_drop: - ALL labels: - "autoheal=true" healthcheck: test: ["CMD", "node", "-e", "fetch('http://127.0.0.1:3000/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))"] interval: 20s timeout: 5s retries: 5 start_period: 30s logging: driver: json-file options: max-size: "10m" max-file: "3" compress: "true" tmpfs: - /tmp/.cache:noexec,nosuid,size=1g security_opt: - no-new-privileges:true mem_limit: 4g memswap_limit: 4g cpus: 1.0 restart: unless-stopped # --------------------------------------------------------------------------- # Redis — BullMQ job queues + rate limiting (internal, not host-published) # Port 6380 (not 6379) so it can never be confused with spotizerr's redis. # --------------------------------------------------------------------------- firecrawl-redis: image: ${FIRECRAWL_REDIS_IMAGE} container_name: firecrawl-redis networks: - firecrawl-net command: redis-server --bind 0.0.0.0 --port 6380 --maxmemory 256mb --maxmemory-policy noeviction # NOTE: no `cap_drop: ALL` here — the entrypoint must chown the bind # mount on first boot (same reason postgres has no cap_drop). Dropping ALL # strips CAP_CHOWN/DAC_OVERRIDE and makes the root entrypoint fail with EPERM. volumes: - ./firecrawl-data/redis:/data labels: - "autoheal=true" healthcheck: test: ["CMD", "redis-cli", "-p", "6380", "ping"] interval: 10s timeout: 3s retries: 5 start_period: 5s logging: driver: json-file options: max-size: "5m" max-file: "2" compress: "true" security_opt: - no-new-privileges:true mem_limit: 256m cpus: 0.5 restart: unless-stopped # --------------------------------------------------------------------------- # RabbitMQ — NUQ transport (internal, not host-published). Default guest/guest # works cross-container on the bridge (verified); no env file needed. # --------------------------------------------------------------------------- firecrawl-rabbitmq: image: ${FIRECRAWL_RABBITMQ_IMAGE} container_name: firecrawl-rabbitmq networks: - firecrawl-net command: rabbitmq-server # NOTE: no `cap_drop: ALL` — entrypoint chowns the bind mount on boot # (same reason postgres has none). Dropping ALL causes EPERM on chown. volumes: - ./firecrawl-data/rabbitmq:/var/lib/rabbitmq labels: - "autoheal=true" healthcheck: test: ["CMD", "rabbitmq-diagnostics", "-q", "check_running"] interval: 10s timeout: 5s retries: 5 start_period: 20s logging: driver: json-file options: max-size: "5m" max-file: "2" compress: "true" security_opt: - no-new-privileges:true mem_limit: 512m cpus: 0.5 restart: unless-stopped # --------------------------------------------------------------------------- # Postgres (NUQ queue store, pg_cron) — internal, not host-published. # DB must stay named 'postgres' (see .firecrawl.env note: the baked # pg_cron extension in this image is configured for cron.database_name='postgres'). # --------------------------------------------------------------------------- firecrawl-postgres: image: ${FIRECRAWL_POSTGRES_IMAGE} container_name: firecrawl-postgres env_file: - ./env/.firecrawl.env networks: - firecrawl-net volumes: - ./firecrawl-data/postgres:/var/lib/postgresql/data labels: - "autoheal=true" healthcheck: test: ["CMD-SHELL", "pg_isready -U postgres"] interval: 10s timeout: 5s retries: 5 start_period: 20s logging: driver: json-file options: max-size: "10m" max-file: "3" compress: "true" security_opt: - no-new-privileges:true mem_limit: 512m cpus: 0.5 restart: unless-stopped