{
 "slug": "block-ai-training",
 "title": "Block AI training, keep AI search",
 "summary": "Refuse the crawlers that feed model training. Keep the ones that put you in ChatGPT, Claude, Perplexity and Gemini answers.",
 "detail": "The distinction most people actually want, and the one that is easy to get wrong: GPTBot trains, OAI-SearchBot indexes for citation. Blocking both loses you the traffic and gains you nothing extra. Google and Apple have no separate crawler at all — Google-Extended and Applebot-Extended are pure control tokens, so they belong in this file while Googlebot and Applebot must not.",
 "crawler_count": 27,
 "crawlers": [
  "anthropic-ai",
  "applebot-extended",
  "bytespider",
  "claudebot",
  "cohere-training-data-crawler",
  "cotoyogi",
  "facebookbot",
  "factset-spyderbot",
  "google-extended",
  "googleother",
  "googleother-image",
  "googleother-video",
  "gptbot",
  "icc-crawler",
  "isscyberriskcrawler",
  "linguee-bot",
  "meta-externalagent",
  "poseidon-research-crawler",
  "quillbot",
  "reflectionbot",
  "sbintuitionsbot",
  "semrushbot-ocob",
  "sidetrade-indexer-bot",
  "tiktokspider",
  "webzio-extended",
  "yandexadditional",
  "yandexadditionalbot"
 ],
 "robots_txt_url": "https://www.pathwren.workers.dev/robots/block-ai-training.txt",
 "robots_txt": "# AI Crawler Index — policy: block-ai-training\n# Block AI training, keep AI search\n# Refuse the crawlers that feed model training. Keep the ones that put you in ChatGPT, Claude, Perplexity and Gemini answers.\n# Generated 2026-09-03 from https://www.pathwren.workers.dev/policy/block-ai-training.html\n# 27 crawlers named. Paste into robots.txt at your document root.\n\nUser-agent: anthropic-ai   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Applebot-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Bytespider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: ClaudeBot\nDisallow: /\n\nUser-agent: cohere-training-data-crawler\nDisallow: /\n\nUser-agent: Cotoyogi\nDisallow: /\n\nUser-agent: FacebookBot\nDisallow: /\n\nUser-agent: Factset_spyderbot\nDisallow: /\n\nUser-agent: Google-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: GoogleOther\nDisallow: /\n\nUser-agent: GoogleOther-Image\nDisallow: /\n\nUser-agent: GoogleOther-Video\nDisallow: /\n\nUser-agent: GPTBot\nDisallow: /\n\nUser-agent: ICC-Crawler\nDisallow: /\n\nUser-agent: ISSCyberRiskCrawler   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: Linguee Bot   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: meta-externalagent\nDisallow: /\n\nUser-agent: Poseidon Research Crawler\nDisallow: /\n\nUser-agent: QuillBot\nDisallow: /\n\nUser-agent: Reflectionbot\nDisallow: /\n\nUser-agent: SBIntuitionsBot\nDisallow: /\n\nUser-agent: SemrushBot-OCOB\nDisallow: /\n\nUser-agent: Sidetrade indexer bot\nDisallow: /\n\nUser-agent: TikTokSpider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: Webzio-Extended\nDisallow: /\n\nUser-agent: YandexAdditional\nDisallow: /\n\nUser-agent: YandexAdditionalBot\nDisallow: /\n\nUser-agent: *\nAllow: /\n\nSitemap: https://www.pathwren.workers.dev/sitemap.xml\n",
 "generated": "2026-09-03",
 "links": [
  {
   "rel": "self",
   "href": "https://www.pathwren.workers.dev/policy/block-ai-training.json",
   "type": "application/json"
  },
  {
   "rel": "changes",
   "href": "https://www.pathwren.workers.dev/changes.json?since=111",
   "type": "application/json",
   "title": "What changed since your cursor — poll this instead of re-downloading this document",
   "cursor_param": "since",
   "head_cursor": 111,
   "min_poll_seconds": 21600,
   "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag."
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/robots/block-ai-training.txt",
   "type": "text/plain"
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/data/agents.json",
   "type": "application/json",
   "title": "Every crawler record in one file"
  },
  {
   "rel": "service-desc",
   "href": "https://www.pathwren.workers.dev/openapi.json",
   "type": "application/json",
   "title": "Every read endpoint, described formally"
  }
 ]
}