# DataCenterIndex — CRAWL node (BHS64b). Workers reach the data node's stores over the private WireGuard # link dci0 (10.68.0.2 → 10.68.0.1). No public port. /healthz + /metrics of the worker and the node # exporter are published on 10.68.0.2 so Prometheus (data node) can scrape them. # # docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl up -d --remove-orphans # docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl run --rm cli run --dry-run --limit 5 # # Profiles: (none) = one worker with DCI_CRAWL_CONCURRENCY jobs · scale = a second worker (worker-b) name: dci x-logging: &logging driver: json-file options: { max-size: "50m", max-file: "5" } x-worker-image: &worker-image image: dci/worker:${DCI_TAG:-latest} build: context: .. dockerfile: deploy/docker/Dockerfile.worker args: TSX_VERSION: ${TSX_VERSION:-4.23.13} x-worker-env: &worker-env NODE_ENV: production TZ: ${TZ:-America/Toronto} DATABASE_URL: postgres://dci:${POSTGRES_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:5432/dci PG_POOL_MAX: ${PG_POOL_MAX:-8} REDIS_URL: redis://:${REDIS_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:6379/0 CLICKHOUSE_URL: http://${DATA_PRIVATE_IP:-10.68.0.1}:8123 CLICKHOUSE_DB: ${CLICKHOUSE_DB:-dci} CLICKHOUSE_USER: dci CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD} S3_ENDPOINT: http://${DATA_PRIVATE_IP:-10.68.0.1}:9000 S3_BUCKET: ${S3_BUCKET:-dci-raw} S3_ACCESS_KEY: ${S3_ACCESS_KEY} S3_SECRET_KEY: ${S3_SECRET_KEY} S3_REGION: us-east-1 SCRAPFLY_API_KEY: ${SCRAPFLY_API_KEY:-} FIRECRAWL_API_KEY: ${FIRECRAWL_API_KEY:-} DCI_SCRAPFLY_DAILY_BUDGET: ${DCI_SCRAPFLY_DAILY_BUDGET:-400} DCI_FIRECRAWL_DAILY_BUDGET: ${DCI_FIRECRAWL_DAILY_BUDGET:-200} DCI_USER_AGENT: ${DCI_USER_AGENT} DCI_CONFIG_DIR: /app/config/connectors DCI_LOG_LEVEL: ${DCI_LOG_LEVEL:-info} DCI_CLICKHOUSE_OPTIONAL: ${DCI_CLICKHOUSE_OPTIONAL:-1} services: worker: <<: *worker-image restart: unless-stopped command: ["worker"] logging: *logging environment: <<: *worker-env DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6} DCI_MAINTENANCE_CONCURRENCY: ${DCI_MAINTENANCE_CONCURRENCY:-1} WORKER_PORT: "8320" ports: - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8320:8320" dns: [1.1.1.1, 9.9.9.9] mem_limit: ${WORKER_MEM_LIMIT:-12g} cpus: ${WORKER_CPUS:-8} stop_grace_period: 90s # Second worker (profile scale) — same image, different metrics port worker-b: <<: *worker-image profiles: ["scale"] restart: unless-stopped command: ["worker"] logging: *logging environment: <<: *worker-env DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6} DCI_MAINTENANCE_CONCURRENCY: "1" WORKER_PORT: "8320" ports: - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8322:8320" dns: [1.1.1.1, 9.9.9.9] mem_limit: ${WORKER_MEM_LIMIT:-12g} cpus: ${WORKER_CPUS:-8} stop_grace_period: 90s # `compose run --rm cli run --dry-run --limit 5` — runs from the crawl IP, like production fetches cli: <<: *worker-image profiles: ["ops"] restart: "no" command: ["cli", "--help"] environment: <<: *worker-env volumes: - /srv/dci/out:/app/data dns: [1.1.1.1, 9.9.9.9] mem_limit: 8g node-exporter: image: prom/node-exporter:v1.9.1 restart: unless-stopped logging: *logging command: - --path.rootfs=/host - --path.procfs=/host/proc - --path.sysfs=/host/sys - --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc|run|var/lib/docker/.+)($$|/) pid: host ports: - "${CRAWL_PRIVATE_IP:-10.68.0.2}:9100:9100" volumes: - /:/host:ro,rslave mem_limit: 256m