import type { Metadata } from "next"; import Link from "next/link"; import type { ReactNode } from "react"; import { Breadcrumbs } from "@/components/entity/breadcrumbs"; import { SectionNav } from "@/components/entity/section-nav"; import { Page, PageHeader, Section } from "@/components/ui/card"; import { TBody, THead, Table, Td, Th, Tr } from "@/components/ui/table"; import { routes } from "@/lib/routes"; import { pageMetadata } from "@/lib/seo"; export const metadata: Metadata = pageMetadata({ title: "About our crawler", description: "DataCenterIndexBot indexes public data center facility pages, press releases, planning filings and cloud region lists. How it identifies itself, how it behaves and how to opt out.", path: routes.bot(), }); /** Default identity of L1 fetches — packages/connectors/src/fetchers.ts (DEFAULT_UA, overridable via DCI_USER_AGENT). */ const UA = "DataCenterIndexBot/0.1 (+https://www.datacenterindex.io/bot; contact@spboucher.ai)"; const NAV = [ { id: "what", label: "What it is" }, { id: "behaviour", label: "Behaviour" }, { id: "identification", label: "Identification" }, { id: "opt-out", label: "Opt out" }, { id: "contact", label: "Contact" }, ]; const FREQUENCY: Array<[string, string]> = [ ["Facility / site pages", "about weekly"], ["Newsroom, press releases, RSS", "daily"], ["Cloud provider region lists", "daily to weekly"], ["Index / listing pages and sitemaps", "weekly to monthly"], ["Planning filings, registries, datasets", "weekly to monthly, per the source's own cadence"], ]; function P({ children }: { children: ReactNode }) { return
{children}
; } const linkCls = "text-ink underline decoration-hair-2 hover:text-accent"; export default function BotPage() { return (DataCenterIndexBot collects public information about data center infrastructure: operators’ facility and campus pages, press releases and newsrooms, cloud providers’ published region and availability-zone lists, planning and permitting filings, utility and regulator publications, and open datasets and registries such as PeeringDB, OpenStreetMap, Wikidata and the World Bank.
From these pages it extracts facts — a facility name, an address, an IT capacity in MW, an opening date, a status, a certification — and records where each fact was read, when, and with what confidence. The result is the structured, versioned graph published on this site and through the public API, with every value linked back to the page it came from and an attribution line for every source (see the source registry). It does not copy articles or reproduce pages; descriptions are short and attributed.
Disallow rules and Crawl-delay; the file is re-read at least once a day. Both the group for DataCenterIndexBot and the * group are respected.
If-None-Match / If-Modified-Since when the server supports them, and every body is content-hashed so an unchanged page is never re-processed.
Typical revisit frequency by page type — actual schedules are set per source and are usually less frequent:
| Page type | Revisit |
|---|---|
| {t} | {f} |
Direct requests carry this User-Agent header:
{UA}
Requests originate from the infrastructure that hosts the index — the MacLustr cluster and dedicated servers at OVHcloud (Canada and France). Reverse DNS on those addresses is not guaranteed to resolve to a datacenterindex.io name, so please rely on the User-Agent string above, or write to us to confirm a specific address. When a source blocks the bot identity and a browser identity is used instead, the same rate limits apply and the source is listed in the registry.
Add a group for the bot to your robots.txt. The change takes effect at the next robots.txt refresh (within 24 hours); pages already archived are not fetched again.
{"User-agent: DataCenterIndexBot\nDisallow: /"}
To exclude only part of a site, list the paths instead of /. A Crawl-delay directive slows the bot down without blocking it.
You can also email contact@spboucher.ai to have a source removed from the index or to correct a value. Removal requests are honoured within 7 days: the connector is paused and archived documents are deleted. Facts already published (for example that a facility exists at an address, with its capacity) remain in the index with their attribution unless you ask for them to be removed as well; corrections are applied at the source and propagate through the normal pipeline, so the change is visible in the live feed.
DataCenterIndex is built and operated by Simon-Pierre Boucher. Questions about the crawler, data corrections, removal requests, data partnerships or API access: contact@spboucher.ai. See also the methodology for the full crawling and reconciliation rules.