{
 "slug": "block-all-ai",
 "title": "Block every AI crawler",
 "summary": "Training, AI search, user-triggered fetches and corpus builders, all refused. Classic search engines still allowed.",
 "detail": "The maximal AI opt-out that still leaves you in Google and Bing. Understand the price before deploying it: you will not be cited by any assistant, and when a reader explicitly asks ChatGPT or Claude to open your page, they get an error. Note also that Perplexity-User and Bytespider are listed here but documented as not governed by robots.txt, so this file is a statement of intent for those two, not an enforcement mechanism.",
 "crawler_count": 77,
 "crawlers": [
  "ai2bot",
  "ai2bot-dolma",
  "aihitbot",
  "aiwebindex",
  "amazonbot",
  "andibot",
  "anomura",
  "anthropic-ai",
  "applebot-extended",
  "atlassian-bot",
  "awariorssbot",
  "awariosmartbot",
  "bedrockbot",
  "bytespider",
  "ccbot",
  "chatgpt-agent",
  "chatgpt-user",
  "claude-searchbot",
  "claude-user",
  "claude-web",
  "claudebot",
  "cloudflare-autorag",
  "cohere-ai",
  "cohere-training-data-crawler",
  "cotoyogi",
  "diffbot",
  "duckassistbot",
  "echoboxbot",
  "exasearchbot",
  "facebookbot",
  "factset-spyderbot",
  "google-agent",
  "google-cloudvertexbot",
  "google-extended",
  "google-gemininotebook",
  "google-pinpoint",
  "google-read-aloud",
  "googleother",
  "googleother-image",
  "googleother-video",
  "gptbot",
  "icc-crawler",
  "imagesiftbot",
  "img2dataset",
  "isscyberriskcrawler",
  "klaviyoaibot",
  "laiondownloader",
  "linguee-bot",
  "meta-externalagent",
  "meta-externalfetcher",
  "meta-webindexer",
  "mistralai-user",
  "oai-searchbot",
  "omgili",
  "omgilibot",
  "panscient",
  "perplexity-user",
  "perplexitybot",
  "phindbot",
  "poseidon-research-crawler",
  "qualifiedbot",
  "quillbot",
  "reflectionbot",
  "sbintuitionsbot",
  "semrushbot-ocob",
  "shapbot",
  "sidetrade-indexer-bot",
  "terracotta",
  "thinkbot",
  "tiktokspider",
  "velenpublicwebcrawler",
  "webzio-extended",
  "yak",
  "yandexadditional",
  "yandexadditionalbot",
  "yandexcalendar",
  "youbot"
 ],
 "robots_txt_url": "https://www.pathwren.workers.dev/robots/block-all-ai.txt",
 "robots_txt": "# AI Crawler Index — policy: block-all-ai\n# Block every AI crawler\n# Training, AI search, user-triggered fetches and corpus builders, all refused. Classic search engines still allowed.\n# Generated 2026-09-03 from https://www.pathwren.workers.dev/policy/block-all-ai.html\n# 77 crawlers named. Paste into robots.txt at your document root.\n\nUser-agent: AI2Bot\nDisallow: /\n\nUser-agent: Ai2Bot-Dolma\nDisallow: /\n\nUser-agent: aiHitBot\nDisallow: /\n\nUser-agent: AIWebIndex\nDisallow: /\n\nUser-agent: Amazonbot\nDisallow: /\n\nUser-agent: Andibot\nDisallow: /\n\nUser-agent: Anomura\nDisallow: /\n\nUser-agent: anthropic-ai   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Applebot-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: atlassian-bot\nDisallow: /\n\nUser-agent: AwarioRssBot\nDisallow: /\n\nUser-agent: AwarioSmartBot\nDisallow: /\n\nUser-agent: bedrockbot\nDisallow: /\n\nUser-agent: Bytespider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: CCBot\nDisallow: /\n\nUser-agent: ChatGPT-User\nDisallow: /\n\nUser-agent: ChatGPT-User\nDisallow: /\n\nUser-agent: Claude-SearchBot\nDisallow: /\n\nUser-agent: Claude-User\nDisallow: /\n\nUser-agent: Claude-Web   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: ClaudeBot\nDisallow: /\n\nUser-agent: Cloudflare-AutoRAG\nDisallow: /\n\nUser-agent: cohere-ai\nDisallow: /\n\nUser-agent: cohere-training-data-crawler\nDisallow: /\n\nUser-agent: Cotoyogi\nDisallow: /\n\nUser-agent: Diffbot\nDisallow: /\n\nUser-agent: DuckAssistBot\nDisallow: /\n\nUser-agent: EchoboxBot\nDisallow: /\n\nUser-agent: ExaSearchBot\nDisallow: /\n\nUser-agent: FacebookBot\nDisallow: /\n\nUser-agent: Factset_spyderbot\nDisallow: /\n\nUser-agent: Google-Agent   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: Google-CloudVertexBot\nDisallow: /\n\nUser-agent: Google-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Google-GeminiNotebook   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: Google-Pinpoint   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: Google-Read-Aloud   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: GoogleOther\nDisallow: /\n\nUser-agent: GoogleOther-Image\nDisallow: /\n\nUser-agent: GoogleOther-Video\nDisallow: /\n\nUser-agent: GPTBot\nDisallow: /\n\nUser-agent: ICC-Crawler\nDisallow: /\n\nUser-agent: ImagesiftBot\nDisallow: /\n\nUser-agent: img2dataset\nDisallow: /\n\nUser-agent: ISSCyberRiskCrawler   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: KlaviyoAIBot\nDisallow: /\n\nUser-agent: LAIONDownloader   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: Linguee Bot   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: meta-externalagent\nDisallow: /\n\nUser-agent: meta-externalfetcher\nDisallow: /\n\nUser-agent: Meta-WebIndexer\nDisallow: /\n\nUser-agent: MistralAI-User\nDisallow: /\n\nUser-agent: OAI-SearchBot\nDisallow: /\n\nUser-agent: omgili\nDisallow: /\n\nUser-agent: omgilibot\nDisallow: /\n\nUser-agent: panscient.com\nDisallow: /\n\nUser-agent: Perplexity-User   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: PerplexityBot\nDisallow: /\n\nUser-agent: PhindBot\nDisallow: /\n\nUser-agent: Poseidon Research Crawler\nDisallow: /\n\nUser-agent: QualifiedBot\nDisallow: /\n\nUser-agent: QuillBot\nDisallow: /\n\nUser-agent: Reflectionbot\nDisallow: /\n\nUser-agent: SBIntuitionsBot\nDisallow: /\n\nUser-agent: SemrushBot-OCOB\nDisallow: /\n\nUser-agent: ShapBot\nDisallow: /\n\nUser-agent: Sidetrade indexer bot\nDisallow: /\n\nUser-agent: TerraCotta\nDisallow: /\n\nUser-agent: Thinkbot   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: TikTokSpider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: VelenPublicWebCrawler\nDisallow: /\n\nUser-agent: Webzio-Extended\nDisallow: /\n\nUser-agent: YaK\nDisallow: /\n\nUser-agent: YandexAdditional\nDisallow: /\n\nUser-agent: YandexAdditionalBot\nDisallow: /\n\nUser-agent: YandexCalendar\nDisallow: /\n\nUser-agent: YouBot\nDisallow: /\n\nUser-agent: *\nAllow: /\n\nSitemap: https://www.pathwren.workers.dev/sitemap.xml\n",
 "generated": "2026-09-03",
 "links": [
  {
   "rel": "self",
   "href": "https://www.pathwren.workers.dev/policy/block-all-ai.json",
   "type": "application/json"
  },
  {
   "rel": "changes",
   "href": "https://www.pathwren.workers.dev/changes.json?since=111",
   "type": "application/json",
   "title": "What changed since your cursor — poll this instead of re-downloading this document",
   "cursor_param": "since",
   "head_cursor": 111,
   "min_poll_seconds": 21600,
   "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag."
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/robots/block-all-ai.txt",
   "type": "text/plain"
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/data/agents.json",
   "type": "application/json",
   "title": "Every crawler record in one file"
  },
  {
   "rel": "service-desc",
   "href": "https://www.pathwren.workers.dev/openapi.json",
   "type": "application/json",
   "title": "Every read endpoint, described formally"
  }
 ]
}