{
 "category": "ai-search",
 "label": "AI search crawlers",
 "description": "Build the retrieval index an assistant answers and cites from. These are the crawlers that send you traffic; blocking them is the expensive mistake in this space.",
 "count": 21,
 "crawlers": [
  {
   "slug": "aiwebindex",
   "name": "AIWebIndex",
   "operator": "Lyrenth",
   "operator_slug": "lyrenth",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "AIWebIndex",
   "user_agent_substring": "AIWebIndex",
   "user_agent_example": "AIWebIndex",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Builds an index of public pages and serves them to AI agents as extracted readable text, with attribution and a link back. Lyrenth publishes a crawler policy stating it does not train foundation models on what it collects and that it obeys robots.txt.",
   "cost_of_blocking": "Agents reading through this index stop seeing you — including the attribution and link back that make it a referral rather than a summary.",
   "operator_docs": "https://lyrenth.com/crawler-policy",
   "html_url": "https://www.pathwren.workers.dev/crawler/aiwebindex.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/aiwebindex.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "amazonbot",
   "name": "Amazonbot",
   "operator": "Amazon",
   "operator_slug": "amazon",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Amazonbot",
   "user_agent_substring": "Amazonbot",
   "user_agent_example": "Mozilla/5.0 (Linux; Android 6.0.1; Nexus 5X Build/MMB29P) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Mobile Safari/537.36 (compatible; Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Amazon's crawler, feeding Alexa's ability to answer questions from the web and Amazon's own search and assistant products.",
   "cost_of_blocking": "Alexa and Amazon's assistants stop answering from your pages. Verify with reverse DNS to crawl.amazonbot.amazon before trusting the user-agent.",
   "operator_docs": "https://developer.amazon.com/amazonbot",
   "html_url": "https://www.pathwren.workers.dev/crawler/amazonbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/amazonbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "andibot",
   "name": "Andibot",
   "operator": "Andi",
   "operator_slug": "andi",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Andibot",
   "user_agent_substring": "Andibot",
   "user_agent_example": "Andibot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The crawler for Andi, a small generative search assistant that summarises pages rather than listing them.",
   "cost_of_blocking": "You disappear from another assistant's answers. Andi publishes no robots.txt statement.",
   "operator_docs": "https://andisearch.com/",
   "html_url": "https://www.pathwren.workers.dev/crawler/andibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/andibot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "anomura",
   "name": "Anomura",
   "operator": "Direqt",
   "operator_slug": "direqt",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Anomura",
   "user_agent_substring": "Anomura",
   "user_agent_example": "Anomura",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Direqt's search crawler. It indexes the sites of Direqt's own publisher customers so their on-site chatbots can answer from them.",
   "cost_of_blocking": "If you are the publisher, this breaks the assistant you put on your own pages. If you are not, it should not be crawling you.",
   "operator_docs": "https://direqt.ai",
   "html_url": "https://www.pathwren.workers.dev/crawler/anomura.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/anomura.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "atlassian-bot",
   "name": "atlassian-bot",
   "operator": "Atlassian",
   "operator_slug": "atlassian",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "atlassian-bot",
   "user_agent_substring": "atlassian-bot",
   "user_agent_example": "atlassian-bot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes a website so it can be searched and cited by Rovo, Atlassian's generative assistant inside Jira and Confluence. Atlassian's documentation walks a customer through editing robots.txt for it, which is as close to a compliance statement as this list gets.",
   "cost_of_blocking": "Rovo cannot answer from your public documentation. If your customers live inside Atlassian tools, this is a support-deflection block.",
   "operator_docs": "https://support.atlassian.com/organization-administration/docs/connect-custom-website-to-rovo/",
   "html_url": "https://www.pathwren.workers.dev/crawler/atlassian-bot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/atlassian-bot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "bedrockbot",
   "name": "bedrockbot",
   "operator": "Amazon",
   "operator_slug": "amazon",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "bedrockbot",
   "user_agent_substring": "bedrockbot",
   "user_agent_example": "bedrockbot-UUID",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The web crawler an AWS customer points at URLs they chose, to build a knowledge base for a Bedrock application. AWS documents that it respects robots.txt and that the user-agent carries a per-customer suffix, so you can allow or refuse one customer's crawl by naming bedrockbot-UUID.",
   "cost_of_blocking": "Companies building retrieval applications on Bedrock cannot include your pages. This is a RAG block, not a training block: nothing is being trained, but nothing can cite you either.",
   "operator_docs": "https://docs.aws.amazon.com/bedrock/latest/userguide/webcrawl-data-source-connector.html",
   "html_url": "https://www.pathwren.workers.dev/crawler/bedrockbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/bedrockbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "claude-searchbot",
   "name": "Claude-SearchBot",
   "operator": "Anthropic",
   "operator_slug": "anthropic",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Claude-SearchBot",
   "user_agent_substring": "Claude-SearchBot",
   "user_agent_example": "Mozilla/5.0 (compatible; Claude-SearchBot/1.0; +Claude-SearchBot@anthropic.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes pages so Claude's web search can find and cite them. Separate token from the training crawler, so search visibility and training consent are independent decisions.",
   "cost_of_blocking": "You stop appearing in Claude's search results and citations.",
   "operator_docs": "https://support.anthropic.com/en/articles/8896518",
   "html_url": "https://www.pathwren.workers.dev/crawler/claude-searchbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/claude-searchbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "claude-web",
   "name": "Claude-Web",
   "operator": "Anthropic",
   "operator_slug": "anthropic",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Claude-Web",
   "user_agent_substring": "Claude-Web",
   "user_agent_example": "Mozilla/5.0 (compatible; Claude-Web/1.0)",
   "respects_robots_txt": "n-a",
   "respects_robots_txt_label": "control token only — no crawler",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "An earlier Anthropic token for user-facing web access, superseded by Claude-User and Claude-SearchBot. Kept here because it appears in most published robots.txt templates.",
   "cost_of_blocking": "None in practice. Retain the rule; expect no traffic.",
   "operator_docs": "https://support.anthropic.com/en/articles/8896518",
   "html_url": "https://www.pathwren.workers.dev/crawler/claude-web.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/claude-web.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "cloudflare-autorag",
   "name": "Cloudflare-AutoRAG",
   "operator": "Cloudflare",
   "operator_slug": "cloudflare",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Cloudflare-AutoRAG",
   "user_agent_substring": "Cloudflare-AutoRAG",
   "user_agent_example": "Cloudflare-AutoRAG",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The crawler behind Cloudflare's AI Search / AutoRAG, which indexes a website into a retrieval index for an application. Cloudflare's own documentation warns that a bot-blocking rule on your zone will also stop this crawler and tells you to allow-list it.",
   "cost_of_blocking": "Applications built on Cloudflare AI Search cannot retrieve your pages. If you are the one building the index over your own site, blocking it breaks your own product.",
   "operator_docs": "https://developers.cloudflare.com/ai-search/",
   "html_url": "https://www.pathwren.workers.dev/crawler/cloudflare-autorag.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/cloudflare-autorag.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "duckassistbot",
   "name": "DuckAssistBot",
   "operator": "DuckDuckGo",
   "operator_slug": "duckduckgo",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "DuckAssistBot",
   "user_agent_substring": "DuckAssistBot",
   "user_agent_example": "Mozilla/5.0 (compatible; DuckAssistBot/1.0; +https://duckduckgo.com/duckassistbot)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Fetches pages so DuckAssist can generate and cite answers inside DuckDuckGo.",
   "cost_of_blocking": "No DuckAssist answers or citations from your site. Ordinary DuckDuckGo results are unaffected.",
   "operator_docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/duckassistbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/duckassistbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "exasearchbot",
   "name": "ExaSearchBot",
   "operator": "Exa",
   "operator_slug": "exa",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "ExaSearchBot",
   "user_agent_substring": "ExaSearchBot",
   "user_agent_example": "ExaSearchBot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Exa's crawler. It discovers and indexes public pages so they can be retrieved and cited through Exa's search API, which is one of the common retrieval backends behind agent frameworks.",
   "cost_of_blocking": "Agents built on Exa's API stop finding you. Exa publishes no statement about robots.txt compliance, so treat the rule as a request.",
   "operator_docs": "https://exa.ai",
   "html_url": "https://www.pathwren.workers.dev/crawler/exasearchbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/exasearchbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-cloudvertexbot",
   "name": "Google-CloudVertexBot",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Google-CloudVertexBot",
   "user_agent_substring": "Google-CloudVertexBot",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-CloudVertexBot/1.0; +https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "Crawls a site on behalf of a Vertex AI Agent Builder customer who is building an agent over that site. It only visits sites the customer has asked it to.",
   "cost_of_blocking": "Third parties can no longer build Vertex AI agents that read your site. Irrelevant to Google Search.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-cloudvertexbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-cloudvertexbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "klaviyoaibot",
   "name": "KlaviyoAIBot",
   "operator": "Klaviyo",
   "operator_slug": "klaviyo",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "KlaviyoAIBot",
   "user_agent_substring": "KlaviyoAIBot",
   "user_agent_example": "KlaviyoAIBot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Fetches pages from domains a Klaviyo customer has explicitly connected to their own account, to power Klaviyo's Kai customer agent. It is scoped to connected domains rather than the open web.",
   "cost_of_blocking": "If the connected domain is yours, blocking this breaks the agent you configured. If it is not, this bot should not be reaching you at all.",
   "operator_docs": "https://help.klaviyo.com/hc/en-us/articles/40496146232219",
   "html_url": "https://www.pathwren.workers.dev/crawler/klaviyoaibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/klaviyoaibot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "meta-webindexer",
   "name": "Meta-WebIndexer",
   "operator": "Meta",
   "operator_slug": "meta",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Meta-WebIndexer",
   "user_agent_substring": "Meta-WebIndexer",
   "user_agent_example": "Meta-WebIndexer",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Per Meta's crawler documentation, Meta-WebIndexer navigates the web to improve the quality of Meta AI's search results. It is a third Meta token alongside Meta-ExternalAgent and Meta-ExternalFetcher, and the newest of them.",
   "cost_of_blocking": "You leave the index Meta AI answers from across Facebook, Instagram and WhatsApp — the largest assistant install base there is. A robots.txt that names the two older Meta tokens does not cover this one.",
   "operator_docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "html_url": "https://www.pathwren.workers.dev/crawler/meta-webindexer.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/meta-webindexer.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "oai-searchbot",
   "name": "OAI-SearchBot",
   "operator": "OpenAI",
   "operator_slug": "openai",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "OAI-SearchBot",
   "user_agent_substring": "OAI-SearchBot",
   "user_agent_example": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-SearchBot/1.0; +https://openai.com/searchbot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://openai.com/searchbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/openai-searchbot.json",
   "ipv4_prefix_count": 35,
   "ipv6_prefix_count": 0,
   "what_it_is": "Builds the index ChatGPT search answers from. Content it collects is used for retrieval and citation, not for model training.",
   "cost_of_blocking": "High. Blocking this removes you from ChatGPT search results and from the source links ChatGPT shows. This is the single most expensive block on this list for anyone who wants to be cited by an assistant.",
   "operator_docs": "https://platform.openai.com/docs/bots",
   "html_url": "https://www.pathwren.workers.dev/crawler/oai-searchbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/oai-searchbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "perplexitybot",
   "name": "PerplexityBot",
   "operator": "Perplexity",
   "operator_slug": "perplexity",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "PerplexityBot",
   "user_agent_substring": "PerplexityBot",
   "user_agent_example": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Safari/537.36; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://www.perplexity.ai/perplexitybot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/perplexity-bot.json",
   "ipv4_prefix_count": 8,
   "ipv6_prefix_count": 0,
   "what_it_is": "Builds Perplexity's search index. Perplexity is citation-heavy by product design, so inclusion here converts to referral traffic more directly than most AI surfaces.",
   "cost_of_blocking": "You stop being indexed and cited by Perplexity, and lose the referral clicks its citations produce.",
   "operator_docs": "https://docs.perplexity.ai/guides/bots",
   "html_url": "https://www.pathwren.workers.dev/crawler/perplexitybot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/perplexitybot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "phindbot",
   "name": "PhindBot",
   "operator": "Phind",
   "operator_slug": "phind",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "PhindBot",
   "user_agent_substring": "PhindBot",
   "user_agent_example": "PhindBot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Phind is an answer engine for developers that combines live web search with its own models. This is the crawler behind those answers.",
   "cost_of_blocking": "You stop being cited in answers to technical questions — which, for documentation and reference sites, is the exact audience most worth keeping.",
   "operator_docs": "https://www.phind.com/",
   "html_url": "https://www.pathwren.workers.dev/crawler/phindbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/phindbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "qualifiedbot",
   "name": "QualifiedBot",
   "operator": "Qualified",
   "operator_slug": "qualified",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "QualifiedBot",
   "user_agent_substring": "QualifiedBot",
   "user_agent_example": "QualifiedBot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Analyses a customer's website so Qualified's AI sales chatbots can answer questions about it in context.",
   "cost_of_blocking": "A chatbot on a site that licensed the product loses context. If that site is yours, this block is self-inflicted.",
   "operator_docs": "https://www.qualified.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/qualifiedbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/qualifiedbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "shapbot",
   "name": "ShapBot",
   "operator": "Parallel",
   "operator_slug": "parallel",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "ShapBot",
   "user_agent_substring": "ShapBot",
   "user_agent_example": "ShapBot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Parallel's crawler. It collects and structures web content to power the search, extraction and deep-research APIs that Parallel sells to agent builders.",
   "cost_of_blocking": "Agents using Parallel's research API lose you as a source. Parallel documents robots.txt compliance, so a rule works.",
   "operator_docs": "https://docs.parallel.ai/features/crawler",
   "html_url": "https://www.pathwren.workers.dev/crawler/shapbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/shapbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "terracotta",
   "name": "TerraCotta",
   "operator": "Ceramic AI",
   "operator_slug": "ceramic",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "TerraCotta",
   "user_agent_substring": "TerraCotta",
   "user_agent_example": "TerraCotta",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Ceramic AI's crawler, which indexes public content for a web-scale search API aimed at LLMs and agents.",
   "cost_of_blocking": "You are absent from another agent-facing retrieval index. Ceramic documents that it obeys robots.txt.",
   "operator_docs": "https://ceramic.ai/",
   "html_url": "https://www.pathwren.workers.dev/crawler/terracotta.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/terracotta.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "youbot",
   "name": "YouBot",
   "operator": "You.com",
   "operator_slug": "you",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "YouBot",
   "user_agent_substring": "YouBot",
   "user_agent_example": "Mozilla/5.0 (compatible; YouBot (+http://www.you.com))",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "You.com's crawler, feeding its AI search product and its search API.",
   "cost_of_blocking": "Removal from You.com's index and from answers built on its API.",
   "operator_docs": "https://about.you.com/youbot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/youbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/youbot.json",
   "last_reviewed": "2026-09-03"
  }
 ]
}