{
 "slug": "maximum-ai-visibility",
 "title": "Maximum AI visibility",
 "summary": "Allow every AI crawler and every search engine; refuse only SEO scrapers. For sites whose goal is to be found and cited by machines.",
 "detail": "If your content exists to be read by assistants — documentation, reference data, an API — every block costs you and none of them protect anything. Pair this with an llms.txt, a sitemap, and per-item JSON, and the crawlers can actually use what they find.",
 "crawler_count": 133,
 "crawlers": [
  "adsbot-google",
  "adsbot-google-mobile",
  "adsbot-google-mobile-apps",
  "ai2bot",
  "ai2bot-dolma",
  "aihitbot",
  "aiwebindex",
  "amazonbot",
  "andibot",
  "anomura",
  "anthropic-ai",
  "apis-google",
  "applebot",
  "applebot-extended",
  "archive-org-bot",
  "atlassian-bot",
  "awariorssbot",
  "awariosmartbot",
  "baiduspider",
  "bedrockbot",
  "bingbot",
  "bytespider",
  "ccbot",
  "chatgpt-agent",
  "chatgpt-user",
  "claude-searchbot",
  "claude-user",
  "claude-web",
  "claudebot",
  "cloudflare-autorag",
  "cohere-ai",
  "cohere-training-data-crawler",
  "cotoyogi",
  "crawl4ai",
  "crawlspace",
  "diffbot",
  "duckassistbot",
  "duckduckbot",
  "echoboxbot",
  "exasearchbot",
  "facebookbot",
  "facebookexternalhit",
  "factset-spyderbot",
  "feedfetcher-google",
  "firecrawlagent",
  "google-agent",
  "google-cloudvertexbot",
  "google-cws",
  "google-extended",
  "google-gemininotebook",
  "google-inspectiontool",
  "google-pinpoint",
  "google-read-aloud",
  "google-safety",
  "google-site-verification",
  "googlebot",
  "googlebot-image",
  "googlebot-news",
  "googlebot-video",
  "googlemessages",
  "googleother",
  "googleother-image",
  "googleother-video",
  "googleproducer",
  "gptbot",
  "ia-archiver",
  "icc-crawler",
  "imagesiftbot",
  "img2dataset",
  "isscyberriskcrawler",
  "kagibot",
  "klaviyoaibot",
  "laiondownloader",
  "lightpanda",
  "linguee-bot",
  "mediapartners-google",
  "meta-externalagent",
  "meta-externalfetcher",
  "meta-webindexer",
  "mistralai-user",
  "mojeekbot",
  "oai-searchbot",
  "omgili",
  "omgilibot",
  "panscient",
  "perplexity-user",
  "perplexitybot",
  "petalbot",
  "phindbot",
  "pinterestbot",
  "poseidon-research-crawler",
  "qualifiedbot",
  "quillbot",
  "qwantbot",
  "qwantbot-news",
  "reflectionbot",
  "sbintuitionsbot",
  "scrapy",
  "screaming-frog-seo-spider",
  "semrushbot-ocob",
  "seznambot",
  "shapbot",
  "sidetrade-indexer-bot",
  "slackbot",
  "slackbot-linkexpanding",
  "storebot-google",
  "terracotta",
  "thinkbot",
  "tiktokspider",
  "timpibot",
  "velenpublicwebcrawler",
  "webzio-extended",
  "wpbot",
  "yak",
  "yandexadditional",
  "yandexadditionalbot",
  "yandexblogs",
  "yandexbot",
  "yandexcalendar",
  "yandexcombot",
  "yandexdirect",
  "yandexfavicons",
  "yandeximages",
  "yandexmarket",
  "yandexmedia",
  "yandexmetrika",
  "yandexmobilebot",
  "yandexrenderresourcesbot",
  "yandexscreenshotbot",
  "yandexvideo",
  "yandexwebmaster",
  "yeti",
  "youbot"
 ],
 "robots_txt_url": "https://www.pathwren.workers.dev/robots/maximum-ai-visibility.txt",
 "robots_txt": "# AI Crawler Index — policy: maximum-ai-visibility\n# Maximum AI visibility\n# Allow every AI crawler and every search engine; refuse only SEO scrapers. For sites whose goal is to be found and cited by machines.\n# Generated 2026-09-03 from https://www.pathwren.workers.dev/policy/maximum-ai-visibility.html\n# 133 crawlers named. Paste into robots.txt at your document root.\n\nUser-agent: AdsBot-Google\nAllow: /\n\nUser-agent: AdsBot-Google-Mobile\nAllow: /\n\nUser-agent: AdsBot-Google-Mobile-Apps\nAllow: /\n\nUser-agent: AI2Bot\nAllow: /\n\nUser-agent: Ai2Bot-Dolma\nAllow: /\n\nUser-agent: aiHitBot\nAllow: /\n\nUser-agent: AIWebIndex\nAllow: /\n\nUser-agent: Amazonbot\nAllow: /\n\nUser-agent: Andibot\nAllow: /\n\nUser-agent: Anomura\nAllow: /\n\nUser-agent: anthropic-ai   # control token, no crawler uses this user-agent\nAllow: /\n\nUser-agent: APIs-Google\nAllow: /\n\nUser-agent: Applebot\nAllow: /\n\nUser-agent: Applebot-Extended   # control token, no crawler uses this user-agent\nAllow: /\n\nUser-agent: archive.org_bot\nAllow: /\n\nUser-agent: atlassian-bot\nAllow: /\n\nUser-agent: AwarioRssBot\nAllow: /\n\nUser-agent: AwarioSmartBot\nAllow: /\n\nUser-agent: Baiduspider\nAllow: /\n\nUser-agent: bedrockbot\nAllow: /\n\nUser-agent: bingbot\nAllow: /\n\nUser-agent: Bytespider   # compliance disputed; enforce at the edge\nAllow: /\n\nUser-agent: CCBot\nAllow: /\n\nUser-agent: ChatGPT-User\nAllow: /\n\nUser-agent: ChatGPT-User\nAllow: /\n\nUser-agent: Claude-SearchBot\nAllow: /\n\nUser-agent: Claude-User\nAllow: /\n\nUser-agent: Claude-Web   # control token, no crawler uses this user-agent\nAllow: /\n\nUser-agent: ClaudeBot\nAllow: /\n\nUser-agent: Cloudflare-AutoRAG\nAllow: /\n\nUser-agent: cohere-ai\nAllow: /\n\nUser-agent: cohere-training-data-crawler\nAllow: /\n\nUser-agent: Cotoyogi\nAllow: /\n\nUser-agent: Crawl4AI\nAllow: /\n\nUser-agent: Crawlspace\nAllow: /\n\nUser-agent: Diffbot\nAllow: /\n\nUser-agent: DuckAssistBot\nAllow: /\n\nUser-agent: DuckDuckBot\nAllow: /\n\nUser-agent: EchoboxBot\nAllow: /\n\nUser-agent: ExaSearchBot\nAllow: /\n\nUser-agent: FacebookBot\nAllow: /\n\nUser-agent: facebookexternalhit\nAllow: /\n\nUser-agent: Factset_spyderbot\nAllow: /\n\nUser-agent: FeedFetcher-Google   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: FirecrawlAgent\nAllow: /\n\nUser-agent: Google-Agent   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-CloudVertexBot\nAllow: /\n\nUser-agent: Google-CWS   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Extended   # control token, no crawler uses this user-agent\nAllow: /\n\nUser-agent: Google-GeminiNotebook   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-InspectionTool\nAllow: /\n\nUser-agent: Google-Pinpoint   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Read-Aloud   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Safety   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Google-Site-Verification   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Googlebot\nAllow: /\n\nUser-agent: Googlebot-Image\nAllow: /\n\nUser-agent: Googlebot-News\nAllow: /\n\nUser-agent: Googlebot-Video\nAllow: /\n\nUser-agent: GoogleMessages   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: GoogleOther\nAllow: /\n\nUser-agent: GoogleOther-Image\nAllow: /\n\nUser-agent: GoogleOther-Video\nAllow: /\n\nUser-agent: GoogleProducer   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: GPTBot\nAllow: /\n\nUser-agent: ia_archiver\nAllow: /\n\nUser-agent: ICC-Crawler\nAllow: /\n\nUser-agent: ImagesiftBot\nAllow: /\n\nUser-agent: img2dataset\nAllow: /\n\nUser-agent: ISSCyberRiskCrawler   # compliance disputed; enforce at the edge\nAllow: /\n\nUser-agent: Kagibot\nAllow: /\n\nUser-agent: KlaviyoAIBot\nAllow: /\n\nUser-agent: LAIONDownloader   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: Lightpanda\nAllow: /\n\nUser-agent: Linguee Bot   # compliance disputed; enforce at the edge\nAllow: /\n\nUser-agent: Mediapartners-Google\nAllow: /\n\nUser-agent: meta-externalagent\nAllow: /\n\nUser-agent: meta-externalfetcher\nAllow: /\n\nUser-agent: Meta-WebIndexer\nAllow: /\n\nUser-agent: MistralAI-User\nAllow: /\n\nUser-agent: MojeekBot\nAllow: /\n\nUser-agent: OAI-SearchBot\nAllow: /\n\nUser-agent: omgili\nAllow: /\n\nUser-agent: omgilibot\nAllow: /\n\nUser-agent: panscient.com\nAllow: /\n\nUser-agent: Perplexity-User   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: PerplexityBot\nAllow: /\n\nUser-agent: PetalBot\nAllow: /\n\nUser-agent: PhindBot\nAllow: /\n\nUser-agent: Pinterestbot\nAllow: /\n\nUser-agent: Poseidon Research Crawler\nAllow: /\n\nUser-agent: QualifiedBot\nAllow: /\n\nUser-agent: QuillBot\nAllow: /\n\nUser-agent: Qwantbot\nAllow: /\n\nUser-agent: Qwantbot-news\nAllow: /\n\nUser-agent: Reflectionbot\nAllow: /\n\nUser-agent: SBIntuitionsBot\nAllow: /\n\nUser-agent: Scrapy\nAllow: /\n\nUser-agent: Screaming Frog SEO Spider\nAllow: /\n\nUser-agent: SemrushBot-OCOB\nAllow: /\n\nUser-agent: SeznamBot\nAllow: /\n\nUser-agent: ShapBot\nAllow: /\n\nUser-agent: Sidetrade indexer bot\nAllow: /\n\nUser-agent: Slackbot\nAllow: /\n\nUser-agent: Slackbot-LinkExpanding\nAllow: /\n\nUser-agent: Storebot-Google\nAllow: /\n\nUser-agent: TerraCotta\nAllow: /\n\nUser-agent: Thinkbot   # compliance disputed; enforce at the edge\nAllow: /\n\nUser-agent: TikTokSpider   # compliance disputed; enforce at the edge\nAllow: /\n\nUser-agent: Timpibot\nAllow: /\n\nUser-agent: VelenPublicWebCrawler\nAllow: /\n\nUser-agent: Webzio-Extended\nAllow: /\n\nUser-agent: wpbot\nAllow: /\n\nUser-agent: YaK\nAllow: /\n\nUser-agent: YandexAdditional\nAllow: /\n\nUser-agent: YandexAdditionalBot\nAllow: /\n\nUser-agent: YandexBlogs\nAllow: /\n\nUser-agent: YandexBot\nAllow: /\n\nUser-agent: YandexCalendar\nAllow: /\n\nUser-agent: YandexComBot\nAllow: /\n\nUser-agent: YandexDirect\nAllow: /\n\nUser-agent: YandexFavicons\nAllow: /\n\nUser-agent: YandexImages\nAllow: /\n\nUser-agent: YandexMarket\nAllow: /\n\nUser-agent: YandexMedia\nAllow: /\n\nUser-agent: YandexMetrika   # operator states robots.txt does not apply; enforce at the edge\nAllow: /\n\nUser-agent: YandexMobileBot\nAllow: /\n\nUser-agent: YandexRenderResourcesBot\nAllow: /\n\nUser-agent: YandexScreenshotBot\nAllow: /\n\nUser-agent: YandexVideo\nAllow: /\n\nUser-agent: YandexWebmaster\nAllow: /\n\nUser-agent: Yeti\nAllow: /\n\nUser-agent: YouBot\nAllow: /\n\nUser-agent: *\nAllow: /\n\nSitemap: https://www.pathwren.workers.dev/sitemap.xml\n",
 "generated": "2026-09-03",
 "links": [
  {
   "rel": "self",
   "href": "https://www.pathwren.workers.dev/policy/maximum-ai-visibility.json",
   "type": "application/json"
  },
  {
   "rel": "changes",
   "href": "https://www.pathwren.workers.dev/changes.json?since=111",
   "type": "application/json",
   "title": "What changed since your cursor — poll this instead of re-downloading this document",
   "cursor_param": "since",
   "head_cursor": 111,
   "min_poll_seconds": 21600,
   "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag."
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/robots/maximum-ai-visibility.txt",
   "type": "text/plain"
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/data/agents.json",
   "type": "application/json",
   "title": "Every crawler record in one file"
  },
  {
   "rel": "service-desc",
   "href": "https://www.pathwren.workers.dev/openapi.json",
   "type": "application/json",
   "title": "Every read endpoint, described formally"
  }
 ]
}