This commit is contained in:
2026-09-26 01:08:50 -07:00
parent fba0546eff
commit 9599867d24
10 changed files with 264 additions and 69 deletions
+233
View File
@@ -0,0 +1,233 @@
# =============================================================================
# Firecrawl — self-hosted web scraper for Hermes (scrape/extract) + vLLM (AI)
# Runs entirely on a private `firecrawl-net` bridge; only the API port
# (FIRECRAWL_PORT=3002) is published to the host. The homepage's host :3000 is
# untouched: playwright binds 3000 INTERNALLY only (firecrawl-net), and the
# api's in-container workers bind the api container's own namespace — so
# nothing firecrawl runs can collide with host :3000. No docker volumes: all
# state lives under ./firecrawl-data/<service>. Images are pulled (never
# built) — see the FIRECRAWL_*_IMAGE digest pins in .env.
#
# Port layout:
# host :3002 -> api container :3002 (the ONLY published port)
# container-internal (firecrawl-net):
# api (express) :3002
# playwright-service :3000
# extract-worker :3004
# queue-worker (liveness) :3005
# nuq-worker x5 :3006-3010
# nuq-prefetch-worker :3011
# nuq-reconciler-worker :3012
# cclog-worker :3013
# postgres :5432 / rabbitmq :5672 / redis :6380 — internal, not published
#
# Env: each firecrawl service loads ONLY its own env file — NEVER the global
# stack .env (it would leak spotizerr's REDIS_PASSWORD into firecrawl-redis,
# whose entrypoint turns it into requirepass and break the passwordless
# REDIS_URL). .firecrawl.env carries the api+postgres contract; the two keys
# needing ${} interpolation (OPENAI_BASE_URL -> vLLM on the LAN,
# SEARXNG_ENDPOINT -> SearXNG on the LAN) live in `environment:` because
# env_file does not interpolate. LAN reach: extra_hosts host-gateway.
#
# Key ordering per service (matches compose.yml):
# image → container_name → env_file → networks/network_mode
# → depends_on → cap_add → ports → volumes → environment
# → labels → healthcheck → security_opt → mem_limit → cpus
# → devices → restart
# =============================================================================
networks:
firecrawl-net:
driver: bridge
services:
# ---------------------------------------------------------------------------
# API + in-container workers (api, queue-worker, extract-worker, nuq workers)
# ---------------------------------------------------------------------------
firecrawl-api:
image: ${FIRECRAWL_IMAGE}
container_name: firecrawl-api
env_file:
- ./env/.firecrawl.env
networks:
- firecrawl-net
depends_on:
firecrawl-postgres:
condition: service_healthy
firecrawl-rabbitmq:
condition: service_healthy
firecrawl-redis:
condition: service_healthy
firecrawl-playwright:
condition: service_healthy
ulimits:
nofile:
soft: 65535
hard: 65535
extra_hosts:
- "host.docker.internal:host-gateway"
ports:
- "${FIRECRAWL_PORT}:3002"
# ${} expansion (env_file does not interpolate): vLLM is host-networked, so
# LAN IP + VLLM_PORT reach it from inside; same for SearXNG.
environment:
- "OPENAI_BASE_URL=http://${LOCAL_IPV4}:${VLLM_PORT}/v1"
- "SEARXNG_ENDPOINT=http://${LOCAL_IPV4}:${SEARXNG_PORT}"
labels:
- "autoheal=true"
healthcheck:
test: ["CMD", "curl", "-sf", "http://127.0.0.1:3002/v0/health/liveness"]
interval: 15s
timeout: 5s
retries: 5
start_period: 60s
logging:
driver: json-file
options:
max-size: "10m"
max-file: "3"
compress: "true"
security_opt:
- no-new-privileges:true
mem_limit: 6g
memswap_limit: 8g
cpus: 2.0
restart: unless-stopped
# ---------------------------------------------------------------------------
# Playwright browser microservice — internal only (port 3000 on firecrawl-net)
# ---------------------------------------------------------------------------
firecrawl-playwright:
image: ${FIRECRAWL_PLAYWRIGHT_IMAGE}
container_name: firecrawl-playwright
env_file:
- ./env/.firecrawl-playwright.env
networks:
- firecrawl-net
cap_drop:
- ALL
labels:
- "autoheal=true"
healthcheck:
test: ["CMD", "node", "-e", "fetch('http://127.0.0.1:3000/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))"]
interval: 20s
timeout: 5s
retries: 5
start_period: 30s
logging:
driver: json-file
options:
max-size: "10m"
max-file: "3"
compress: "true"
tmpfs:
- /tmp/.cache:noexec,nosuid,size=1g
security_opt:
- no-new-privileges:true
mem_limit: 4g
memswap_limit: 4g
cpus: 1.0
restart: unless-stopped
# ---------------------------------------------------------------------------
# Redis — BullMQ job queues + rate limiting (internal, not host-published)
# Port 6380 (not 6379) so it can never be confused with spotizerr's redis.
# ---------------------------------------------------------------------------
firecrawl-redis:
image: ${FIRECRAWL_REDIS_IMAGE}
container_name: firecrawl-redis
networks:
- firecrawl-net
command: redis-server --bind 0.0.0.0 --port 6380 --maxmemory 256mb --maxmemory-policy noeviction
# NOTE: no `cap_drop: ALL` here — the entrypoint must chown the bind
# mount on first boot (same reason postgres has no cap_drop). Dropping ALL
# strips CAP_CHOWN/DAC_OVERRIDE and makes the root entrypoint fail with EPERM.
volumes:
- ./firecrawl-data/redis:/data
labels:
- "autoheal=true"
healthcheck:
test: ["CMD", "redis-cli", "-p", "6380", "ping"]
interval: 10s
timeout: 3s
retries: 5
start_period: 5s
logging:
driver: json-file
options:
max-size: "5m"
max-file: "2"
compress: "true"
security_opt:
- no-new-privileges:true
mem_limit: 256m
cpus: 0.5
restart: unless-stopped
# ---------------------------------------------------------------------------
# RabbitMQ — NUQ transport (internal, not host-published). Default guest/guest
# works cross-container on the bridge (verified); no env file needed.
# ---------------------------------------------------------------------------
firecrawl-rabbitmq:
image: ${FIRECRAWL_RABBITMQ_IMAGE}
container_name: firecrawl-rabbitmq
networks:
- firecrawl-net
command: rabbitmq-server
# NOTE: no `cap_drop: ALL` — entrypoint chowns the bind mount on boot
# (same reason postgres has none). Dropping ALL causes EPERM on chown.
volumes:
- ./firecrawl-data/rabbitmq:/var/lib/rabbitmq
labels:
- "autoheal=true"
healthcheck:
test: ["CMD", "rabbitmq-diagnostics", "-q", "check_running"]
interval: 10s
timeout: 5s
retries: 5
start_period: 20s
logging:
driver: json-file
options:
max-size: "5m"
max-file: "2"
compress: "true"
security_opt:
- no-new-privileges:true
mem_limit: 512m
cpus: 0.5
restart: unless-stopped
# ---------------------------------------------------------------------------
# Postgres (NUQ queue store, pg_cron) — internal, not host-published.
# DB must stay named 'postgres' (see .firecrawl.env note: the baked
# pg_cron extension in this image is configured for cron.database_name='postgres').
# ---------------------------------------------------------------------------
firecrawl-postgres:
image: ${FIRECRAWL_POSTGRES_IMAGE}
container_name: firecrawl-postgres
env_file:
- ./env/.firecrawl.env
networks:
- firecrawl-net
volumes:
- ./firecrawl-data/postgres:/var/lib/postgresql/data
labels:
- "autoheal=true"
healthcheck:
test: ["CMD-SHELL", "pg_isready -U postgres"]
interval: 10s
timeout: 5s
retries: 5
start_period: 20s
logging:
driver: json-file
options:
max-size: "10m"
max-file: "3"
compress: "true"
security_opt:
- no-new-privileges:true
mem_limit: 512m
cpus: 0.5
restart: unless-stopped