{
 "slug": "allow-ai-search-only",
 "title": "Allow AI search and user fetches, block the rest",
 "summary": "Be findable and citable in assistants without contributing to training corpora.",
 "detail": "The inverse framing of block-ai-training, written as an allowlist so the default for anything new is deny. Fetches a user explicitly asked for stay allowed, because refusing those produces a visible error for a real person who wanted your page.",
 "crawler_count": 65,
 "crawlers": [
  "aiwebindex",
  "amazonbot",
  "andibot",
  "anomura",
  "applebot",
  "atlassian-bot",
  "baiduspider",
  "bedrockbot",
  "bingbot",
  "chatgpt-agent",
  "chatgpt-user",
  "claude-searchbot",
  "claude-user",
  "claude-web",
  "cloudflare-autorag",
  "cohere-ai",
  "duckassistbot",
  "duckduckbot",
  "exasearchbot",
  "facebookexternalhit",
  "google-agent",
  "google-cloudvertexbot",
  "google-gemininotebook",
  "google-pinpoint",
  "google-read-aloud",
  "googlebot",
  "googlebot-image",
  "googlebot-news",
  "googlebot-video",
  "googlemessages",
  "kagibot",
  "klaviyoaibot",
  "meta-externalfetcher",
  "meta-webindexer",
  "mistralai-user",
  "mojeekbot",
  "oai-searchbot",
  "perplexity-user",
  "perplexitybot",
  "petalbot",
  "phindbot",
  "pinterestbot",
  "qualifiedbot",
  "qwantbot",
  "qwantbot-news",
  "seznambot",
  "shapbot",
  "slackbot",
  "slackbot-linkexpanding",
  "storebot-google",
  "terracotta",
  "timpibot",
  "yandexblogs",
  "yandexbot",
  "yandexcalendar",
  "yandexcombot",
  "yandexfavicons",
  "yandeximages",
  "yandexmarket",
  "yandexmedia",
  "yandexmobilebot",
  "yandexrenderresourcesbot",
  "yandexvideo",
  "yeti",
  "youbot"
 ],
 "robots_txt_url": "https://www.pathwren.workers.dev/robots/allow-ai-search-only.txt",
 "robots_txt": "# AI Crawler Index — policy: allow-ai-search-only\n# Allow AI search and user fetches, block the rest\n# Be findable and citable in assistants without contributing to training corpora.\n# Generated 2026-09-03 from https://www.pathwren.workers.dev/policy/allow-ai-search-only.html\n# 65 crawlers named. Paste into robots.txt at your document root.\n\nUser-agent: AIWebIndex\nAllow: /\n\nUser-agent: Amazonbot\nAllow: /\n\nUser-agent: Andibot\nAllow: /\n\nUser-agent: Anomura\nAllow: /\n\nUser-agent: Applebot\nAllow: /\n\nUser-agent: atlassian-bot\nAllow: /\n\nUser-agent: Baiduspider\nAllow: /\n\nUser-agent: bedrockbot\nAllow: /\n\nUser-agent: bingbot\nAllow: /\n\nUser-agent: ChatGPT-User\nAllow: /\n\nUser-agent: ChatGPT-User\nAllow: /\n\nUser-agent: Claude-SearchBot\nAllow: /\n\nUser-agent: Claude-User\nAllow: /\n\nUser-agent: Claude-Web   # control token, no crawler uses this user-agent\nAllow: /\n\nUser-agent: Cloudflare-AutoRAG\nAllow: /\n\nUser-agent: cohere-ai\nAllow: /\n\nUser-agent: DuckAssistBot\nAllow: /\n\nUser-agent: DuckDuckBot\nAllow: /\n\nUser-agent: ExaSearchBot\nAllow: /\n\nUser-agent: facebookexternalhit\nAllow: /\n\nUser-agent: Google-Agent   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-CloudVertexBot\nAllow: /\n\nUser-agent: Google-GeminiNotebook   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Pinpoint   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Read-Aloud   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Googlebot\nAllow: /\n\nUser-agent: Googlebot-Image\nAllow: /\n\nUser-agent: Googlebot-News\nAllow: /\n\nUser-agent: Googlebot-Video\nAllow: /\n\nUser-agent: GoogleMessages   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Kagibot\nAllow: /\n\nUser-agent: KlaviyoAIBot\nAllow: /\n\nUser-agent: meta-externalfetcher\nAllow: /\n\nUser-agent: Meta-WebIndexer\nAllow: /\n\nUser-agent: MistralAI-User\nAllow: /\n\nUser-agent: MojeekBot\nAllow: /\n\nUser-agent: OAI-SearchBot\nAllow: /\n\nUser-agent: Perplexity-User   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: PerplexityBot\nAllow: /\n\nUser-agent: PetalBot\nAllow: /\n\nUser-agent: PhindBot\nAllow: /\n\nUser-agent: Pinterestbot\nAllow: /\n\nUser-agent: QualifiedBot\nAllow: /\n\nUser-agent: Qwantbot\nAllow: /\n\nUser-agent: Qwantbot-news\nAllow: /\n\nUser-agent: SeznamBot\nAllow: /\n\nUser-agent: ShapBot\nAllow: /\n\nUser-agent: Slackbot\nAllow: /\n\nUser-agent: Slackbot-LinkExpanding\nAllow: /\n\nUser-agent: Storebot-Google\nAllow: /\n\nUser-agent: TerraCotta\nAllow: /\n\nUser-agent: Timpibot\nAllow: /\n\nUser-agent: YandexBlogs\nAllow: /\n\nUser-agent: YandexBot\nAllow: /\n\nUser-agent: YandexCalendar\nAllow: /\n\nUser-agent: YandexComBot\nAllow: /\n\nUser-agent: YandexFavicons\nAllow: /\n\nUser-agent: YandexImages\nAllow: /\n\nUser-agent: YandexMarket\nAllow: /\n\nUser-agent: YandexMedia\nAllow: /\n\nUser-agent: YandexMobileBot\nAllow: /\n\nUser-agent: YandexRenderResourcesBot\nAllow: /\n\nUser-agent: YandexVideo\nAllow: /\n\nUser-agent: Yeti\nAllow: /\n\nUser-agent: YouBot\nAllow: /\n\n# Anything not named above is refused.\nUser-agent: *\nDisallow: /\n\nSitemap: https://www.pathwren.workers.dev/sitemap.xml\n",
 "generated": "2026-09-03",
 "links": [
  {
   "rel": "self",
   "href": "https://www.pathwren.workers.dev/policy/allow-ai-search-only.json",
   "type": "application/json"
  },
  {
   "rel": "changes",
   "href": "https://www.pathwren.workers.dev/changes.json?since=111",
   "type": "application/json",
   "title": "What changed since your cursor — poll this instead of re-downloading this document",
   "cursor_param": "since",
   "head_cursor": 111,
   "min_poll_seconds": 21600,
   "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag."
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/robots/allow-ai-search-only.txt",
   "type": "text/plain"
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/data/agents.json",
   "type": "application/json",
   "title": "Every crawler record in one file"
  },
  {
   "rel": "service-desc",
   "href": "https://www.pathwren.workers.dev/openapi.json",
   "type": "application/json",
   "title": "Every read endpoint, described formally"
  }
 ]
}