{
 "category": "dataset",
 "label": "Corpus and dataset builders",
 "description": "Crawl the web into a published or resold dataset that other people train on. Highest leverage per block, longest delay before any effect.",
 "count": 17,
 "crawlers": [
  {
   "slug": "ai2bot",
   "name": "AI2Bot",
   "operator": "Allen Institute for AI",
   "operator_slug": "ai2",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "AI2Bot",
   "user_agent_substring": "AI2Bot",
   "user_agent_example": "Mozilla/5.0 (compatible) AI2Bot (+https://www.allenai.org/crawler)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The Allen Institute's crawler, gathering pages for open research corpora such as Dolma that underpin fully open models like OLMo.",
   "cost_of_blocking": "Excluded from open research datasets. Worth a deliberate decision: this is the category where 'blocking AI' also blocks the open, auditable end of it.",
   "operator_docs": "https://allenai.org/crawler",
   "html_url": "https://www.pathwren.workers.dev/crawler/ai2bot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/ai2bot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "ai2bot-dolma",
   "name": "Ai2Bot-Dolma",
   "operator": "Allen Institute for AI",
   "operator_slug": "ai2",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "Ai2Bot-Dolma",
   "user_agent_substring": "Ai2Bot-Dolma",
   "user_agent_example": "Mozilla/5.0 (compatible) Ai2Bot-Dolma (+https://www.allenai.org/crawler)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The variant of AI2's crawler named for the Dolma corpus specifically.",
   "cost_of_blocking": "Same as AI2Bot: exclusion from an open, published training corpus.",
   "operator_docs": "https://allenai.org/crawler",
   "html_url": "https://www.pathwren.workers.dev/crawler/ai2bot-dolma.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/ai2bot-dolma.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "aihitbot",
   "name": "aiHitBot",
   "operator": "aiHit",
   "operator_slug": "aihit",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "aiHitBot",
   "user_agent_substring": "aiHitBot",
   "user_agent_example": "aiHitBot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "aiHit's automated collector, building a company dataset from public company websites.",
   "cost_of_blocking": "Your company record in a B2B dataset goes stale. Documented as respecting robots.txt, so the rule works.",
   "operator_docs": "https://www.aihitdata.com/about",
   "html_url": "https://www.pathwren.workers.dev/crawler/aihitbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/aihitbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "awariorssbot",
   "name": "AwarioRssBot",
   "operator": "Awario",
   "operator_slug": "awario",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "AwarioRssBot",
   "user_agent_substring": "AwarioRssBot",
   "user_agent_example": "AwarioRssBot/1.0 (+https://awario.com/bots.html; bots@awario.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The feed-reading half of Awario's pair, documented on the same page and under the same crawl-rate policy.",
   "cost_of_blocking": "Your RSS updates stop reaching Awario's monitoring. Block both tokens or neither.",
   "operator_docs": "https://awario.com/bots.html",
   "html_url": "https://www.pathwren.workers.dev/crawler/awariorssbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/awariorssbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "awariosmartbot",
   "name": "AwarioSmartBot",
   "operator": "Awario",
   "operator_slug": "awario",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "AwarioSmartBot",
   "user_agent_substring": "AwarioSmartBot",
   "user_agent_example": "AwarioSmartBot/1.0 (+https://awario.com/bots.html; bots@awario.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Awario's brand-monitoring crawler. It documents one request per three seconds, honours Crawl-delay, and states it does not use consecutive IP blocks so identification is by user-agent only.",
   "cost_of_blocking": "Mentions of brands on your pages stop being surfaced to the people monitoring them — including, quite possibly, your own.",
   "operator_docs": "https://awario.com/bots.html",
   "html_url": "https://www.pathwren.workers.dev/crawler/awariosmartbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/awariosmartbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "ccbot",
   "name": "CCBot",
   "operator": "Common Crawl",
   "operator_slug": "commoncrawl",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "CCBot",
   "user_agent_substring": "CCBot",
   "user_agent_example": "CCBot/2.0 (https://commoncrawl.org/faq/)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://index.commoncrawl.org/ccbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/commoncrawl-ccbot.json",
   "ipv4_prefix_count": 4,
   "ipv6_prefix_count": 1,
   "what_it_is": "Common Crawl's corpus builder. It trains nothing itself, but its archive is an input to most open and many closed LLM training sets, which makes it the highest-leverage single entry on this list.",
   "cost_of_blocking": "Future Common Crawl snapshots exclude you, so downstream training sets lose you too — but only going forward. Existing snapshots are permanent and blocking today does not retract them.",
   "operator_docs": "https://commoncrawl.org/ccbot",
   "html_url": "https://www.pathwren.workers.dev/crawler/ccbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/ccbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "diffbot",
   "name": "Diffbot",
   "operator": "Diffbot",
   "operator_slug": "diffbot",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "Diffbot",
   "user_agent_substring": "Diffbot",
   "user_agent_example": "Mozilla/5.0 (compatible; Diffbot/0.1; +http://www.diffbot.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Extracts structured records from pages to build a commercial knowledge graph that is resold and used for retrieval and training.",
   "cost_of_blocking": "Your facts stop entering a widely-licensed knowledge graph. Whether that is a loss depends on whether you want to be a machine-readable entity.",
   "operator_docs": "https://docs.diffbot.com/",
   "html_url": "https://www.pathwren.workers.dev/crawler/diffbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/diffbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "echoboxbot",
   "name": "EchoboxBot",
   "operator": "Echobox",
   "operator_slug": "echobox",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "EchoboxBot",
   "user_agent_substring": "EchoboxBot",
   "user_agent_example": "EchoboxBot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Collects data supporting Echobox's AI-driven social and email distribution products, which publishers use to schedule and target their own content.",
   "cost_of_blocking": "Publishers using Echobox get worse scheduling decisions about your articles. No compliance statement is published.",
   "operator_docs": "https://echobox.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/echoboxbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/echoboxbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "imagesiftbot",
   "name": "ImagesiftBot",
   "operator": "Hive AI",
   "operator_slug": "hive",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "ImagesiftBot",
   "user_agent_substring": "ImagesiftBot",
   "user_agent_example": "Mozilla/5.0 (compatible; ImagesiftBot; +imagesift.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Crawls images for Hive AI's reverse-image and dataset products. Image-heavy sites see this one long before they see the text crawlers.",
   "cost_of_blocking": "Your images stop entering an image dataset and reverse-image index.",
   "operator_docs": "https://imagesift.com/about",
   "html_url": "https://www.pathwren.workers.dev/crawler/imagesiftbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/imagesiftbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "img2dataset",
   "name": "img2dataset",
   "operator": "LAION / img2dataset",
   "operator_slug": "laion",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "img2dataset",
   "user_agent_substring": "img2dataset",
   "user_agent_example": "img2dataset",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The tool used to turn image-URL lists such as LAION's into downloaded training sets. It is run by whoever is building a dataset, not by a single operator.",
   "cost_of_blocking": "Your images are skipped when someone materialises an image-text dataset that references them.",
   "operator_docs": "https://github.com/rom1504/img2dataset",
   "html_url": "https://www.pathwren.workers.dev/crawler/img2dataset.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/img2dataset.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "laiondownloader",
   "name": "LAIONDownloader",
   "operator": "LAION / img2dataset",
   "operator_slug": "laion",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "LAIONDownloader",
   "user_agent_substring": "LAIONDownloader",
   "user_agent_example": "LAIONDownloader",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "LAION's downloader, used to materialise the image and text datasets the non-profit publishes for machine-learning research. LAION's own FAQ is the source for its robots.txt position.",
   "cost_of_blocking": "Your media is skipped when an open research dataset is built from URL lists. Once a dataset is published, a later block does not remove you from it.",
   "operator_docs": "https://laion.ai/faq/",
   "html_url": "https://www.pathwren.workers.dev/crawler/laiondownloader.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/laiondownloader.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "omgili",
   "name": "omgili",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "omgili",
   "user_agent_substring": "omgili",
   "user_agent_example": "omgili/0.5 +http://omgili.com",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The older robots token for the same Webz.io collection, still honoured and still worth listing.",
   "cost_of_blocking": "Same as omgilibot.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/omgili.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/omgili.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "omgilibot",
   "name": "omgilibot",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "omgilibot",
   "user_agent_substring": "omgilibot",
   "user_agent_example": "omgilibot/0.4; +http://omgili.com",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Webz.io's crawler, collecting web and forum text sold as datasets, including to model builders.",
   "cost_of_blocking": "Exclusion from a commercial dataset resold to third parties.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/omgilibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/omgilibot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "panscient",
   "name": "Panscient",
   "operator": "Panscient",
   "operator_slug": "panscient",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "panscient.com",
   "user_agent_substring": "panscient.com",
   "user_agent_example": "Mozilla/5.0 (compatible; panscient.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Compiles structured data about businesses and business professionals using machine learning. Panscient's FAQ states it obeys robots.txt.",
   "cost_of_blocking": "Your company pages stop feeding a business-data product. No effect on search or assistants.",
   "operator_docs": "https://panscient.com/faq.htm",
   "html_url": "https://www.pathwren.workers.dev/crawler/panscient.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/panscient.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "thinkbot",
   "name": "Thinkbot",
   "operator": "Thinkbot",
   "operator_slug": "thinkbot",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "Thinkbot",
   "user_agent_substring": "Thinkbot",
   "user_agent_example": "Thinkbot",
   "respects_robots_txt": "disputed",
   "respects_robots_txt_label": "compliance disputed",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Collects pages for analysis of how sites are adopting AI and automation. The ai.robots.txt dataset records the operator as not respecting robots.txt.",
   "cost_of_blocking": "Exclusion from a market-research dataset. Expect to enforce this at the edge rather than in robots.txt.",
   "operator_docs": "https://www.thinkbot.agency",
   "html_url": "https://www.pathwren.workers.dev/crawler/thinkbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/thinkbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "velenpublicwebcrawler",
   "name": "VelenPublicWebCrawler",
   "operator": "Hunter (Velen)",
   "operator_slug": "hunter",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "VelenPublicWebCrawler",
   "user_agent_substring": "VelenPublicWebCrawler",
   "user_agent_example": "VelenPublicWebCrawler",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Hunter's crawler, written in Go, building business datasets and machine-learning models from public pages. Its page states it follows robots.txt and meta directives and never fetches more than one page every two seconds.",
   "cost_of_blocking": "Your company pages stop feeding a B2B contact and company dataset. The crawl rate it documents makes this one of the cheapest visitors to simply allow.",
   "operator_docs": "https://velen.io/",
   "html_url": "https://www.pathwren.workers.dev/crawler/velenpublicwebcrawler.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/velenpublicwebcrawler.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yak",
   "name": "YaK",
   "operator": "Meltwater",
   "operator_slug": "meltwater",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "YaK",
   "user_agent_substring": "YaK",
   "user_agent_example": "YaK",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Meltwater's crawler, feeding the live data stream behind its media-monitoring and consumer-intelligence suite.",
   "cost_of_blocking": "Your content stops appearing in Meltwater's media monitoring — which is how PR teams find out you were mentioned. Some publishers want to be in it.",
   "operator_docs": "https://www.meltwater.com/en/suite/consumer-intelligence",
   "html_url": "https://www.pathwren.workers.dev/crawler/yak.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yak.json",
   "last_reviewed": "2026-09-03"
  }
 ]
}