spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1# DataCenterIndex — CRAWL node (BHS64b). Workers reach the data node's stores over the private WireGuard2# link dci0 (10.68.0.2 → 10.68.0.1). No public port. /healthz + /metrics of the worker and the node3# exporter are published on 10.68.0.2 so Prometheus (data node) can scrape them.4#5# docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl up -d --remove-orphans6# docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl run --rm cli run <connector> --dry-run --limit 57#8# Profiles: (none) = one worker with DCI_CRAWL_CONCURRENCY jobs · scale = a second worker (worker-b)9name: dci1011x-logging: &logging12 driver: json-file13 options: { max-size: "50m", max-file: "5" }1415x-worker-image: &worker-image16 image: dci/worker:${DCI_TAG:-latest}17 build:18 context: ..19 dockerfile: deploy/docker/Dockerfile.worker20 args:21 TSX_VERSION: ${TSX_VERSION:-4.23.13}2223x-worker-env: &worker-env24 NODE_ENV: production25 TZ: ${TZ:-America/Toronto}26 DATABASE_URL: postgres://dci:${POSTGRES_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:5432/dci27 PG_POOL_MAX: ${PG_POOL_MAX:-8}28 REDIS_URL: redis://:${REDIS_PASSWORD}@${DATA_PRIVATE_IP:-10.68.0.1}:6379/029 CLICKHOUSE_URL: http://${DATA_PRIVATE_IP:-10.68.0.1}:812330 CLICKHOUSE_DB: ${CLICKHOUSE_DB:-dci}31 CLICKHOUSE_USER: dci32 CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD}33 S3_ENDPOINT: http://${DATA_PRIVATE_IP:-10.68.0.1}:900034 S3_BUCKET: ${S3_BUCKET:-dci-raw}35 S3_ACCESS_KEY: ${S3_ACCESS_KEY}36 S3_SECRET_KEY: ${S3_SECRET_KEY}37 S3_REGION: us-east-138 SCRAPFLY_API_KEY: ${SCRAPFLY_API_KEY:-}39 FIRECRAWL_API_KEY: ${FIRECRAWL_API_KEY:-}40 DCI_SCRAPFLY_DAILY_BUDGET: ${DCI_SCRAPFLY_DAILY_BUDGET:-400}41 DCI_FIRECRAWL_DAILY_BUDGET: ${DCI_FIRECRAWL_DAILY_BUDGET:-200}42 DCI_USER_AGENT: ${DCI_USER_AGENT}43 DCI_CONFIG_DIR: /app/config/connectors44 DCI_LOG_LEVEL: ${DCI_LOG_LEVEL:-info}45 DCI_CLICKHOUSE_OPTIONAL: ${DCI_CLICKHOUSE_OPTIONAL:-1}4647services:48 worker:49 <<: *worker-image50 restart: unless-stopped51 command: ["worker"]52 logging: *logging53 environment:54 <<: *worker-env55 DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6}56 DCI_MAINTENANCE_CONCURRENCY: ${DCI_MAINTENANCE_CONCURRENCY:-1}57 WORKER_PORT: "8320"58 ports:59 - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8320:8320"60 dns: [1.1.1.1, 9.9.9.9]61 mem_limit: ${WORKER_MEM_LIMIT:-12g}62 cpus: ${WORKER_CPUS:-8}63 stop_grace_period: 90s6465 # Second worker (profile scale) — same image, different metrics port66 worker-b:67 <<: *worker-image68 profiles: ["scale"]69 restart: unless-stopped70 command: ["worker"]71 logging: *logging72 environment:73 <<: *worker-env74 DCI_CRAWL_CONCURRENCY: ${DCI_CRAWL_CONCURRENCY:-6}75 DCI_MAINTENANCE_CONCURRENCY: "1"76 WORKER_PORT: "8320"77 ports:78 - "${CRAWL_PRIVATE_IP:-10.68.0.2}:8322:8320"79 dns: [1.1.1.1, 9.9.9.9]80 mem_limit: ${WORKER_MEM_LIMIT:-12g}81 cpus: ${WORKER_CPUS:-8}82 stop_grace_period: 90s8384 # `compose run --rm cli run <id> --dry-run --limit 5` — runs from the crawl IP, like production fetches85 cli:86 <<: *worker-image87 profiles: ["ops"]88 restart: "no"89 command: ["cli", "--help"]90 environment:91 <<: *worker-env92 volumes:93 - /srv/dci/out:/app/data94 dns: [1.1.1.1, 9.9.9.9]95 mem_limit: 8g9697 node-exporter:98 image: prom/node-exporter:v1.9.199 restart: unless-stopped100 logging: *logging101 command:102 - --path.rootfs=/host103 - --path.procfs=/host/proc104 - --path.sysfs=/host/sys105 - --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc|run|var/lib/docker/.+)($$|/)106 pid: host107 ports:108 - "${CRAWL_PRIVATE_IP:-10.68.0.2}:9100:9100"109 volumes:110 - /:/host:ro,rslave111 mem_limit: 256m112