id: meta-news name: Meta data center news domain: about.fb.com kind: news mode: rss priority: 2 license: "Publicly published corporate newsroom feeds; headlines, dates and facts only, attribution kept" attribution: "Meta — about.fb.com/news ; datacenters.atmeta.com/blog" homepage: https://about.fb.com/news/ coverage: { global: true, operators: ["Meta"] } implementation: cloud_newsroom fetch: level: 1 maxLevel: 2 rpm: 10 concurrency: 1 respectRobots: true maxCreditsPerRun: 0 discovery: rss: - "https://about.fb.com/feed/" sitemap: ["https://datacenters.atmeta.com/post-sitemap.xml"] include: ["about\\.fb\\.com/news/20\\d{2}/", "datacenters\\.atmeta\\.com/20\\d{2}/\\d{2}/"] exclude: ["/feed/?$", "\\?share=", "/wp-json/"] classify: - { pattern: "datacenters\\.atmeta\\.com/20\\d{2}/\\d{2}/", group: newsroom, pageType: press_release, priority: 75 } - { pattern: "about\\.fb\\.com/news/20\\d{2}/", group: newsroom, pageType: press_release, priority: 65 } schedule: newsroom: daily rss: daily sitemap: daily params: include: "\\b(data ?cent(er|re)s?|datacenter|AI infrastructure|campus|invest(s|ing|ment)? (of |more than |over )?(\\$|€|£)\\d|(\\$|€|£)\\d+(\\.\\d+)? ?(billion|million)|gigawatt|megawatt|\\bMW\\b|\\bGW\\b|Hyperion|Prometheus|compute cluster|break(s|ing)? ground|(we are|now|comes|is) online|cooling|substation|clean energy|nuclear|geothermal)\\b" exclude: "\\b(Instagram|WhatsApp|Threads|Messenger|Quest|Ray-Ban|Horizon|Reels|creators?|teen accounts?|election)\\b" defaults: operatorName: Meta notes: > about.fb.com/feed (10 items, mostly product news) filtered on title/summary; the Meta Data Centers blog (datacenters.atmeta.com — "Kansas City, we are online!", "Hello, Bowling Green!", grants) has no working RSS (/feed/ returns the HTML home page) but publishes a Yoast post-sitemap.xml (~68 dated posts). discovery.include keeps only dated article URLs so static pages (community-grants, funds) are not turned into news. Both hosts allow crawling.