SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
2 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
3.7 KB · 112 lines yaml
Raw Blame History
1# DataCenterIndex — CRAWL node (BHS64b). Workers reach the data node's stores over the private WireGuard2# link dci0 (10.68.0.2 → 10.68.0.1). No public port. /healthz + /metrics of the worker and the node3# exporter are published on 10.68.0.2 so Prometheus (data node) can scrape them.4#5#   docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl up -d --remove-orphans6#   docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl run --rm cli run <connector> --dry-run --limit 57#8# Profiles: (none) = one worker with DCI_CRAWL_CONCURRENCY jobs · scale = a second worker (worker-b)9name: dci1011x-logging: &logging12  driver: json-file13  options: { max-size: "50m", max-file: "5" }1415x-worker-image: &worker-image16  image: dci/worker:${DCI_TAG:-latest}17  build:18    context: ..19    dockerfile: deploy/docker/Dockerfile.worker20    args:21      TSX_VERSION: ${TSX_VERSION:-4.23.13}2223x-worker-env: &worker-env24  NODE_ENV: production25  TZ: ${TZ:-America/Toronto}26  DATABASE_URL: postgres://dci:${POSTGRES_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:5432/dci27  PG_POOL_MAX: ${PG_POOL_MAX:-8}28  REDIS_URL: redis://:${REDIS_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:6379/029  CLICKHOUSE_URL: http://${DATA_PRIVATE_IP:-10.68.0.1}:812330  CLICKHOUSE_DB: ${CLICKHOUSE_DB:-dci}31  CLICKHOUSE_USER: dci32  CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD}33  S3_ENDPOINT: http://${DATA_PRIVATE_IP:-10.68.0.1}:900034  S3_BUCKET: ${S3_BUCKET:-dci-raw}35  S3_ACCESS_KEY: ${S3_ACCESS_KEY}36  S3_SECRET_KEY: ${S3_SECRET_KEY}37  S3_REGION: us-east-138  SCRAPFLY_API_KEY: ${SCRAPFLY_API_KEY:-}39  FIRECRAWL_API_KEY: ${FIRECRAWL_API_KEY:-}40  DCI_SCRAPFLY_DAILY_BUDGET: ${DCI_SCRAPFLY_DAILY_BUDGET:-400}41  DCI_FIRECRAWL_DAILY_BUDGET: ${DCI_FIRECRAWL_DAILY_BUDGET:-200}42  DCI_USER_AGENT: ${DCI_USER_AGENT}43  DCI_CONFIG_DIR: /app/config/connectors44  DCI_LOG_LEVEL: ${DCI_LOG_LEVEL:-info}45  DCI_CLICKHOUSE_OPTIONAL: ${DCI_CLICKHOUSE_OPTIONAL:-1}4647services:48  worker:49    <<: *worker-image50    restart: unless-stopped51    command: ["worker"]52    logging: *logging53    environment:54      <<: *worker-env55      DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6}56      DCI_MAINTENANCE_CONCURRENCY: ${DCI_MAINTENANCE_CONCURRENCY:-1}57      WORKER_PORT: "8320"58    ports:59      - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8320:8320"60    dns: [1.1.1.1, 9.9.9.9]61    mem_limit: ${WORKER_MEM_LIMIT:-12g}62    cpus: ${WORKER_CPUS:-8}63    stop_grace_period: 90s6465  # Second worker (profile scale) — same image, different metrics port66  worker-b:67    <<: *worker-image68    profiles: ["scale"]69    restart: unless-stopped70    command: ["worker"]71    logging: *logging72    environment:73      <<: *worker-env74      DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6}75      DCI_MAINTENANCE_CONCURRENCY: "1"76      WORKER_PORT: "8320"77    ports:78      - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8322:8320"79    dns: [1.1.1.1, 9.9.9.9]80    mem_limit: ${WORKER_MEM_LIMIT:-12g}81    cpus: ${WORKER_CPUS:-8}82    stop_grace_period: 90s8384  # `compose run --rm cli run <id> --dry-run --limit 5` — runs from the crawl IP, like production fetches85  cli:86    <<: *worker-image87    profiles: ["ops"]88    restart: "no"89    command: ["cli", "--help"]90    environment:91      <<: *worker-env92    volumes:93      - /srv/dci/out:/app/data94    dns: [1.1.1.1, 9.9.9.9]95    mem_limit: 8g9697  node-exporter:98    image: prom/node-exporter:v1.9.199    restart: unless-stopped100    logging: *logging101    command:102      - --path.rootfs=/host103      - --path.procfs=/host/proc104      - --path.sysfs=/host/sys105      - --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc|run|var/lib/docker/.+)($$|/)106    pid: host107    ports:108      - "${CRAWL_PRIVATE_IP:-10.68.0.2}:9100:9100"109    volumes:110      - /:/host:ro,rslave111    mem_limit: 256m112