{
 "category": "ai-training",
 "label": "AI training crawlers",
 "description": "Collect pages in bulk so that a model can be trained or fine-tuned on them. Blocking these removes you from future training sets and changes nothing a user sees today.",
 "count": 27,
 "crawlers": [
  {
   "slug": "anthropic-ai",
   "name": "anthropic-ai",
   "operator": "Anthropic",
   "operator_slug": "anthropic",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "anthropic-ai",
   "user_agent_substring": "anthropic-ai",
   "user_agent_example": "(no live crawler currently identifies with this string)",
   "respects_robots_txt": "n-a",
   "respects_robots_txt_label": "control token only — no crawler",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A legacy robots.txt token from before Anthropic consolidated on ClaudeBot. It is still widely present in robots.txt files and costs nothing to keep, but it is a control token rather than a bot you will see in logs.",
   "cost_of_blocking": "None. Nothing crawls under this name today; keeping the rule is harmless insurance.",
   "operator_docs": "https://support.anthropic.com/en/articles/8896518",
   "html_url": "https://www.pathwren.workers.dev/crawler/anthropic-ai.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/anthropic-ai.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "applebot-extended",
   "name": "Applebot-Extended",
   "operator": "Apple",
   "operator_slug": "apple",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Applebot-Extended",
   "user_agent_substring": "(control token only — no crawler)",
   "user_agent_example": "(none: Applebot-Extended never appears as a user-agent)",
   "respects_robots_txt": "n-a",
   "respects_robots_txt_label": "control token only — no crawler",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Apple's counterpart to Google-Extended: a robots.txt token that withdraws consent for Apple Intelligence and Apple foundation-model training, without touching Applebot's search crawl.",
   "cost_of_blocking": "Excluded from Apple Intelligence training. Siri, Spotlight and Safari suggestions are unaffected.",
   "operator_docs": "https://support.apple.com/en-us/119829",
   "html_url": "https://www.pathwren.workers.dev/crawler/applebot-extended.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/applebot-extended.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "bytespider",
   "name": "Bytespider",
   "operator": "ByteDance",
   "operator_slug": "bytedance",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Bytespider",
   "user_agent_substring": "Bytespider",
   "user_agent_example": "Mozilla/5.0 (Linux; Android 5.0) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; Bytespider; spider-feedback@bytedance.com)",
   "respects_robots_txt": "disputed",
   "respects_robots_txt_label": "compliance disputed",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "ByteDance's crawler, associated with training data collection for Doubao and related models. Repeatedly reported by CDNs and site operators as the highest-volume AI crawler on the web and as inconsistent about robots.txt.",
   "cost_of_blocking": "Little to lose. If you want it gone, expect to block by user-agent at the edge rather than to ask politely in robots.txt.",
   "operator_docs": "https://www.bytespider.net/",
   "html_url": "https://www.pathwren.workers.dev/crawler/bytespider.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/bytespider.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "claudebot",
   "name": "ClaudeBot",
   "operator": "Anthropic",
   "operator_slug": "anthropic",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "ClaudeBot",
   "user_agent_substring": "ClaudeBot",
   "user_agent_example": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Anthropic's bulk crawler, gathering pages that may be used to train Claude models.",
   "cost_of_blocking": "Content excluded from training data for future Claude models. No effect on Claude's ability to fetch a link a user gives it.",
   "operator_docs": "https://support.anthropic.com/en/articles/8896518",
   "html_url": "https://www.pathwren.workers.dev/crawler/claudebot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/claudebot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "cohere-training-data-crawler",
   "name": "cohere-training-data-crawler",
   "operator": "Cohere",
   "operator_slug": "cohere",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "cohere-training-data-crawler",
   "user_agent_substring": "cohere-training-data-crawler",
   "user_agent_example": "Mozilla/5.0 (compatible; cohere-training-data-crawler)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Cohere's separately-named bulk crawler for model training data, split out so consent for training and consent for retrieval can differ.",
   "cost_of_blocking": "Excluded from Cohere model training.",
   "operator_docs": "https://cohere.com/",
   "html_url": "https://www.pathwren.workers.dev/crawler/cohere-training-data-crawler.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/cohere-training-data-crawler.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "cotoyogi",
   "name": "Cotoyogi",
   "operator": "ROIS-DS",
   "operator_slug": "rois",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Cotoyogi",
   "user_agent_substring": "Cotoyogi",
   "user_agent_example": "Cotoyogi",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A crawler run by ROIS-DS, a Japanese inter-university research organisation, collecting Japanese-language text for AI training. It publishes a crawler page in English and Japanese.",
   "cost_of_blocking": "Your Japanese-language content is left out of an academic training corpus.",
   "operator_docs": "https://ds.rois.ac.jp/en_center8/en_crawler/",
   "html_url": "https://www.pathwren.workers.dev/crawler/cotoyogi.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/cotoyogi.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "facebookbot",
   "name": "FacebookBot",
   "operator": "Meta",
   "operator_slug": "meta",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "FacebookBot",
   "user_agent_substring": "FacebookBot",
   "user_agent_example": "FacebookBot/1.0 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Meta's older speech- and language-corpus crawler, largely superseded by meta-externalagent but still listed as a valid robots token.",
   "cost_of_blocking": "Negligible today. Keep the rule; expect little traffic.",
   "operator_docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/facebookbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/facebookbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "factset-spyderbot",
   "name": "Factset_spyderbot",
   "operator": "FactSet",
   "operator_slug": "factset",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Factset_spyderbot",
   "user_agent_substring": "Factset_spyderbot",
   "user_agent_example": "Factset_spyderbot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "FactSet's crawler, collecting data used in AI model training for its financial data and analytics products.",
   "cost_of_blocking": "Exclusion from a financial-data vendor's corpus. Relevant mostly to companies whose filings and disclosures are being read.",
   "operator_docs": "https://www.factset.com/ai",
   "html_url": "https://www.pathwren.workers.dev/crawler/factset-spyderbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/factset-spyderbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-extended",
   "name": "Google-Extended",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Google-Extended",
   "user_agent_substring": "(control token only — no crawler)",
   "user_agent_example": "(none: Google-Extended never appears as a user-agent)",
   "respects_robots_txt": "n-a",
   "respects_robots_txt_label": "control token only — no crawler",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Not a crawler. A robots.txt token that tells Google whether pages Googlebot already fetched may be used to train and ground Gemini. You will never see it in an access log; disallowing it changes what Google does with content it fetched under a different name.",
   "cost_of_blocking": "You are excluded from Gemini grounding and Gemini training. Google Search ranking and indexing are explicitly unaffected. This is the cleanest 'no training, keep my search traffic' lever that exists.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-extended.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-extended.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googleother",
   "name": "GoogleOther",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GoogleOther",
   "user_agent_substring": "GoogleOther",
   "user_agent_example": "Mozilla/5.0 (compatible; GoogleOther)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "A generic fetcher used by Google product teams for one-off crawls and research, including data collection that does not belong to Search.",
   "cost_of_blocking": "No effect on Search indexing. Blocks internal Google research and product fetches.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googleother.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googleother.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googleother-image",
   "name": "GoogleOther-Image",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GoogleOther-Image",
   "user_agent_substring": "GoogleOther-Image",
   "user_agent_example": "GoogleOther-Image/1.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "The image variant of GoogleOther: one-off fetches by Google product and research teams that are not Search. It also answers to a GoogleOther group in robots.txt.",
   "cost_of_blocking": "Google teams outside Search stop fetching your images. Image Search itself is unaffected — that is Googlebot-Image.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googleother-image.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googleother-image.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googleother-video",
   "name": "GoogleOther-Video",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GoogleOther-Video",
   "user_agent_substring": "GoogleOther-Video",
   "user_agent_example": "GoogleOther-Video/1.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "The video variant of GoogleOther, used for internal Google fetches that do not belong to Search.",
   "cost_of_blocking": "No effect on Search or on Google Video search. Blocks internal Google research fetches of your video files.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googleother-video.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googleother-video.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "gptbot",
   "name": "GPTBot",
   "operator": "OpenAI",
   "operator_slug": "openai",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GPTBot",
   "user_agent_substring": "GPTBot",
   "user_agent_example": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.2; +https://openai.com/gptbot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://openai.com/gptbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/openai-gptbot.json",
   "ipv4_prefix_count": 21,
   "ipv6_prefix_count": 0,
   "what_it_is": "OpenAI's bulk crawler. Pages it fetches may be used to train future OpenAI foundation models. It is not the bot that puts you in ChatGPT's search results, and blocking it does not remove you from them.",
   "cost_of_blocking": "Your content is excluded from training data for future OpenAI models. No effect on ChatGPT search visibility, on citations, or on links a user pastes into ChatGPT.",
   "operator_docs": "https://platform.openai.com/docs/bots",
   "html_url": "https://www.pathwren.workers.dev/crawler/gptbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/gptbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "icc-crawler",
   "name": "ICC-Crawler",
   "operator": "NICT",
   "operator_slug": "nict",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "ICC-Crawler",
   "user_agent_substring": "ICC-Crawler",
   "user_agent_example": "ICC-Crawler",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Operated by NICT, Japan's national information and communications research institute. The collected data supports AI research and, per the operator, is also provided to third parties including commercial companies.",
   "cost_of_blocking": "You are excluded from a national research corpus and from the commercial redistributions of it. This is a dataset-shaped block: one refusal, many downstream effects.",
   "operator_docs": "https://www.nict.go.jp/en/",
   "html_url": "https://www.pathwren.workers.dev/crawler/icc-crawler.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/icc-crawler.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "isscyberriskcrawler",
   "name": "ISSCyberRiskCrawler",
   "operator": "ISS Corporate Solutions",
   "operator_slug": "iss",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "ISSCyberRiskCrawler",
   "user_agent_substring": "ISSCyberRiskCrawler",
   "user_agent_example": "ISSCyberRiskCrawler",
   "respects_robots_txt": "disputed",
   "respects_robots_txt_label": "compliance disputed",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Crawls in order to train models that score a company's cyber risk. The ai.robots.txt dataset records the operator as not respecting robots.txt; ISS publishes no compliance statement of its own.",
   "cost_of_blocking": "A rule here is a statement of intent. Your organisation's public footprint still gets scored — by a model trained on everybody else.",
   "operator_docs": "https://iss-cyber.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/isscyberriskcrawler.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/isscyberriskcrawler.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "linguee-bot",
   "name": "Linguee Bot",
   "operator": "Linguee",
   "operator_slug": "linguee",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Linguee Bot",
   "user_agent_substring": "Linguee Bot",
   "user_agent_example": "Linguee Bot",
   "respects_robots_txt": "disputed",
   "respects_robots_txt_label": "compliance disputed",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Gathers bilingual text for Linguee's translation corpus and the machine translation trained on it. Recorded in the ai.robots.txt dataset as not respecting robots.txt.",
   "cost_of_blocking": "Multilingual pages stop feeding a translation corpus. If your site is translated, being in it is usually a benefit.",
   "operator_docs": "https://www.linguee.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/linguee-bot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/linguee-bot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "meta-externalagent",
   "name": "meta-externalagent",
   "operator": "Meta",
   "operator_slug": "meta",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "meta-externalagent",
   "user_agent_substring": "meta-externalagent",
   "user_agent_example": "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Meta's AI crawler, gathering training data for Llama and Meta AI. It replaced the older FacebookBot name for this purpose.",
   "cost_of_blocking": "Excluded from Meta AI training. Link previews on Facebook, Instagram and WhatsApp are unaffected — those are a different bot.",
   "operator_docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/meta-externalagent.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/meta-externalagent.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "poseidon-research-crawler",
   "name": "Poseidon Research Crawler",
   "operator": "Poseidon Research",
   "operator_slug": "poseidon",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Poseidon Research Crawler",
   "user_agent_substring": "Poseidon Research Crawler",
   "user_agent_example": "Poseidon Research Crawler",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A crawler run by Poseidon Research, a lab working on interpretability research for AI systems.",
   "cost_of_blocking": "Exclusion from an interpretability research corpus. No published compliance statement.",
   "operator_docs": "https://www.poseidonresearch.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/poseidon-research-crawler.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/poseidon-research-crawler.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "quillbot",
   "name": "QuillBot",
   "operator": "QuillBot",
   "operator_slug": "quillbot",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "QuillBot",
   "user_agent_substring": "QuillBot",
   "user_agent_example": "QuillBot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Operated by QuillBot as part of its writing, paraphrasing and AI-detection products. The dataset also records a second token, quillbot.com, for the same operator.",
   "cost_of_blocking": "Exclusion from QuillBot's corpus. No compliance statement is published, so the rule is a request.",
   "operator_docs": "https://quillbot.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/quillbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/quillbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "reflectionbot",
   "name": "Reflectionbot",
   "operator": "Reflection AI",
   "operator_slug": "reflection",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Reflectionbot",
   "user_agent_substring": "Reflectionbot",
   "user_agent_example": "Reflectionbot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "An undocumented crawler whose user-agent links to Reflection AI, a company building AI models. The link in the user-agent is the only public statement of purpose that exists.",
   "cost_of_blocking": "Unknown by construction — which is itself the reason some people block it. Nothing user-facing depends on it.",
   "operator_docs": "https://reflection.ai/",
   "html_url": "https://www.pathwren.workers.dev/crawler/reflectionbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/reflectionbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "sbintuitionsbot",
   "name": "SBIntuitionsBot",
   "operator": "SB Intuitions",
   "operator_slug": "sbintuitions",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "SBIntuitionsBot",
   "user_agent_substring": "SBIntuitionsBot",
   "user_agent_example": "SBIntuitionsBot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "SB Intuitions is SoftBank's Japanese LLM lab; this crawler gathers data used in that model development and in information analysis. The operator publishes a dedicated bot page.",
   "cost_of_blocking": "Your content is excluded from a Japanese-language foundation-model corpus. Nothing user-facing changes.",
   "operator_docs": "https://www.sbintuitions.co.jp/en/bot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/sbintuitionsbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/sbintuitionsbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "semrushbot-ocob",
   "name": "SemrushBot-OCOB",
   "operator": "Semrush",
   "operator_slug": "semrush",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "SemrushBot-OCOB",
   "user_agent_substring": "SemrushBot-OCOB",
   "user_agent_example": "Mozilla/5.0 (compatible; SemrushBot-OCOB/1.0; +http://www.semrush.com/bot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Semrush's separately-tokenised crawler for its AI content tooling, split out so SEO crawling and AI reuse can be answered differently.",
   "cost_of_blocking": "Exclusion from Semrush's AI corpus, with its SEO crawl unaffected.",
   "operator_docs": "https://www.semrush.com/bot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/semrushbot-ocob.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/semrushbot-ocob.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "sidetrade-indexer-bot",
   "name": "Sidetrade indexer bot",
   "operator": "Sidetrade",
   "operator_slug": "sidetrade",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Sidetrade indexer bot",
   "user_agent_substring": "Sidetrade indexer bot",
   "user_agent_example": "Sidetrade indexer bot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Sidetrade extracts web data for a range of uses including training its AI products for order-to-cash and customer-data work.",
   "cost_of_blocking": "Exclusion from a commercial B2B dataset. The operator publishes no robots.txt statement, so treat the rule as a request rather than a control.",
   "operator_docs": "https://www.sidetrade.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/sidetrade-indexer-bot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/sidetrade-indexer-bot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "tiktokspider",
   "name": "TikTokSpider",
   "operator": "ByteDance",
   "operator_slug": "bytedance",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "TikTokSpider",
   "user_agent_substring": "TikTokSpider",
   "user_agent_example": "Mozilla/5.0 (compatible; TikTokSpider; ttspider-feedback@tiktok.com)",
   "respects_robots_txt": "disputed",
   "respects_robots_txt_label": "compliance disputed",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A second ByteDance crawler identifying with TikTok, collecting page content for the same family of models.",
   "cost_of_blocking": "Little to lose unless TikTok search referral matters to you.",
   "operator_docs": "https://www.bytespider.net/",
   "html_url": "https://www.pathwren.workers.dev/crawler/tiktokspider.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/tiktokspider.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "webzio-extended",
   "name": "Webzio-Extended",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Webzio-Extended",
   "user_agent_substring": "Webzio-Extended",
   "user_agent_example": "Mozilla/5.0 (compatible; Webzio-Extended/1.0)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Webz.io's opt-out token specifically for AI training reuse, in the pattern Google and Apple established.",
   "cost_of_blocking": "Your content is excluded from the AI-training tier of Webz.io's product while ordinary collection continues.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/webzio-extended.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/webzio-extended.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexadditional",
   "name": "YandexAdditional",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "YandexAdditional",
   "user_agent_substring": "YandexAdditional",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexAdditional/1.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The token that controls whether already-indexed pages may appear in Search with Yandex AI answers. Yandex's table says it makes no indexing requests of its own — it exists so a site can opt out of the generative answer without leaving the index.",
   "cost_of_blocking": "You disappear from Yandex's AI answers while staying in Yandex Search. This is Yandex's equivalent of Google-Extended, and it is the cheap opt-out most people are looking for.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexadditional.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexadditional.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexadditionalbot",
   "name": "YandexAdditionalBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "YandexAdditionalBot",
   "user_agent_substring": "YandexAdditionalBot",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexAdditionalBot/1.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The second token Yandex publishes for the same AI-answers opt-out. Both names appear in Yandex's own robot list, so a robots.txt that names only one of them is half a policy.",
   "cost_of_blocking": "Same as YandexAdditional: out of Yandex's AI answers, still in Yandex Search. Name both tokens or neither.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexadditionalbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexadditionalbot.json",
   "last_reviewed": "2026-09-03"
  }
 ]
}