{
 "category": "search",
 "label": "Search engines",
 "description": "Classic index-and-rank crawlers. Several also feed their operator's generative answers, which is why the AI opt-out for Google and Apple is a token rather than a block.",
 "count": 28,
 "crawlers": [
  {
   "slug": "applebot",
   "name": "Applebot",
   "operator": "Apple",
   "operator_slug": "apple",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Applebot",
   "user_agent_substring": "Applebot",
   "user_agent_example": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/13.1.1 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://search.developer.apple.com/applebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/apple-applebot.json",
   "ipv4_prefix_count": 33,
   "ipv6_prefix_count": 0,
   "what_it_is": "Powers Siri, Spotlight and Safari suggestions. Blocking it is a search decision, not an AI decision — the AI decision has its own token.",
   "cost_of_blocking": "You disappear from Siri, Spotlight and Safari search suggestions across Apple's install base.",
   "operator_docs": "https://support.apple.com/en-us/119829",
   "html_url": "https://www.pathwren.workers.dev/crawler/applebot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/applebot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "baiduspider",
   "name": "Baiduspider",
   "operator": "Baidu",
   "operator_slug": "baidu",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Baiduspider",
   "user_agent_substring": "Baiduspider",
   "user_agent_example": "Mozilla/5.0 (compatible; Baiduspider/2.0; +http://www.baidu.com/search/spider.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Baidu's search crawler, and the ingest path for Baidu's Ernie-backed answers.",
   "cost_of_blocking": "Removal from Baidu Search, which matters only if you want Chinese-language traffic.",
   "operator_docs": "https://help.baidu.com/question?prod_id=99&class=0&id=3001",
   "html_url": "https://www.pathwren.workers.dev/crawler/baiduspider.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/baiduspider.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "bingbot",
   "name": "bingbot",
   "operator": "Microsoft",
   "operator_slug": "microsoft",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "bingbot",
   "user_agent_substring": "bingbot",
   "user_agent_example": "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://www.bing.com/toolbox/bingbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/bing-bingbot.json",
   "ipv4_prefix_count": 28,
   "ipv6_prefix_count": 0,
   "what_it_is": "Bing's only crawler, and therefore also the crawler behind Microsoft Copilot's grounding. Microsoft's documented way to keep search indexing while refusing generative reuse is the nocache / noarchive robots meta directive, not a separate user-agent.",
   "cost_of_blocking": "Very high and very wide: Bing, Copilot, DuckDuckGo and several assistants that resell Bing's index all lose you at once. Use nocache/noarchive rather than blocking.",
   "operator_docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
   "html_url": "https://www.pathwren.workers.dev/crawler/bingbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/bingbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "duckduckbot",
   "name": "DuckDuckBot",
   "operator": "DuckDuckGo",
   "operator_slug": "duckduckgo",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "DuckDuckBot",
   "user_agent_substring": "DuckDuckBot",
   "user_agent_example": "DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://duckduckgo.com/duckduckbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/duckduckgo-duckduckbot.json",
   "ipv4_prefix_count": 486,
   "ipv6_prefix_count": 0,
   "what_it_is": "DuckDuckGo's own crawler. Note that the bulk of DuckDuckGo's web results come from Bing, so blocking bingbot removes you from DuckDuckGo whether or not you allow this one.",
   "cost_of_blocking": "Limited on its own; the real DuckDuckGo lever is bingbot.",
   "operator_docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/duckduckbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/duckduckbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googlebot",
   "name": "Googlebot",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot",
   "user_agent_substring": "Googlebot",
   "user_agent_example": "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 170,
   "ipv6_prefix_count": 147,
   "what_it_is": "The classic search crawler. It is also the crawler behind AI Overviews: Google does not run a separate bot for them, which is why the only AI opt-out is the Google-Extended token and not a Googlebot block.",
   "cost_of_blocking": "Total. You leave Google Search. Never block this to avoid AI use; use Google-Extended instead.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googlebot-image",
   "name": "Googlebot-Image",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot-Image",
   "user_agent_substring": "Googlebot-Image",
   "user_agent_example": "Googlebot-Image/1.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 170,
   "ipv6_prefix_count": 147,
   "what_it_is": "Image indexing for Google Images. A separate token so you can leave images out of search without leaving search.",
   "cost_of_blocking": "Your images stop appearing in Google Images.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot-image.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot-image.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googlebot-news",
   "name": "Googlebot-News",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot-News",
   "user_agent_substring": "Googlebot-News",
   "user_agent_example": "(uses the Googlebot user-agent; controlled by the Googlebot-News robots token)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 170,
   "ipv6_prefix_count": 147,
   "what_it_is": "A robots.txt token controlling inclusion in Google News. It does not have its own user-agent string; the fetch arrives as Googlebot.",
   "cost_of_blocking": "Removal from Google News, with normal Search unaffected.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot-news.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot-news.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googlebot-video",
   "name": "Googlebot-Video",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot-Video",
   "user_agent_substring": "Googlebot-Video",
   "user_agent_example": "Googlebot-Video/1.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 170,
   "ipv6_prefix_count": 147,
   "what_it_is": "The video half of Googlebot. It crawls video files and the pages around them for Google Video search, and it is matched by a robots.txt group for Googlebot as well as by its own token.",
   "cost_of_blocking": "Your videos leave Google video search. A rule for Googlebot already covers it, so blocking this token alone is usually a mistake of precision rather than of intent.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot-video.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot-video.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "kagibot",
   "name": "Kagibot",
   "operator": "Kagi",
   "operator_slug": "kagi",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Kagibot",
   "user_agent_substring": "Kagibot",
   "user_agent_example": "Mozilla/5.0 (compatible; Kagibot/1.0; +https://kagi.com/bot)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The crawler for Kagi, a paid, ad-free search engine with its own index and its own assistant.",
   "cost_of_blocking": "You leave Kagi's index. Kagi's users are paying to search and skew technical; per visitor this is an expensive block.",
   "operator_docs": "https://kagi.com/bot",
   "html_url": "https://www.pathwren.workers.dev/crawler/kagibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/kagibot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "mojeekbot",
   "name": "MojeekBot",
   "operator": "Mojeek",
   "operator_slug": "mojeek",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "MojeekBot",
   "user_agent_substring": "MojeekBot",
   "user_agent_example": "Mozilla/5.0 (compatible; MojeekBot/0.11; +https://www.mojeek.com/bot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Mojeek's crawler. Mojeek runs one of the few genuinely independent web indexes — not a front end over Bing or Google — so it is one of the few blocks that removes you from an index nobody else can put you back into. Its documentation states it obeys the first record whose User-Agent contains MojeekBot, falling back to *.",
   "cost_of_blocking": "You leave an independent index that other privacy-focused search products draw on. Small traffic, disproportionate long-term cost to web plurality.",
   "operator_docs": "https://www.mojeek.com/bot.html",
   "html_url": "https://www.pathwren.workers.dev/crawler/mojeekbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/mojeekbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "petalbot",
   "name": "PetalBot",
   "operator": "Huawei",
   "operator_slug": "huawei",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "PetalBot",
   "user_agent_substring": "PetalBot",
   "user_agent_example": "Mozilla/5.0 (Linux; Android 7.0;) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; PetalBot;+https://webmaster.petalsearch.com/site/petalbot)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Huawei's crawler for Petal Search, shipped as the default search on Huawei devices.",
   "cost_of_blocking": "Removal from Petal Search. Frequently blocked for volume rather than for policy.",
   "operator_docs": "https://aspiegel.com/petalbot",
   "html_url": "https://www.pathwren.workers.dev/crawler/petalbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/petalbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "pinterestbot",
   "name": "Pinterestbot",
   "operator": "Pinterest",
   "operator_slug": "pinterest",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Pinterestbot",
   "user_agent_substring": "Pinterestbot",
   "user_agent_example": "Mozilla/5.0 (compatible; Pinterestbot/1.0; +https://www.pinterest.com/bot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Pinterest's crawler. It indexes pages so people can find them on Pinterest and re-reads product pages to keep price and title on a Pin current. Pinterest states that content it crawls is not used to train their Canvas image generation model.",
   "cost_of_blocking": "Pins pointing at your site go stale — wrong prices, dead links — and new content stops being indexed. For a retailer this is one of the more expensive blocks on the list.",
   "operator_docs": "https://help.pinterest.com/en/business/article/pinterest-crawler",
   "html_url": "https://www.pathwren.workers.dev/crawler/pinterestbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/pinterestbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "qwantbot",
   "name": "Qwantbot",
   "operator": "Qwant",
   "operator_slug": "qwant",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Qwantbot",
   "user_agent_substring": "Qwantbot",
   "user_agent_example": "Mozilla/5.0 (compatible; Qwantbot/1.0_12345; +https://help.qwant.com/bot/)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Qwant's crawler. Qwant documents that the string Qwantbot always appears in its user-agents whatever the crawler version, which is what makes a substring match safe here.",
   "cost_of_blocking": "You leave the index behind Qwant, the French privacy-focused engine, and the products that federate it.",
   "operator_docs": "https://help.qwant.com/bot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/qwantbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/qwantbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "qwantbot-news",
   "name": "Qwantbot-news",
   "operator": "Qwant",
   "operator_slug": "qwant",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Qwantbot-news",
   "user_agent_substring": "Qwantbot-news",
   "user_agent_example": "Mozilla/5.0 (compatible; Qwantbot-news/2.0; +https://help.qwant.com/bot/)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The news variant of Qwant's crawler, documented alongside the main one and carrying the same Qwantbot substring.",
   "cost_of_blocking": "Your articles stop appearing in Qwant News. A rule for Qwantbot as a substring already catches both.",
   "operator_docs": "https://help.qwant.com/bot/",
   "html_url": "https://www.pathwren.workers.dev/crawler/qwantbot-news.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/qwantbot-news.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "seznambot",
   "name": "SeznamBot",
   "operator": "Seznam",
   "operator_slug": "seznam",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "SeznamBot",
   "user_agent_substring": "SeznamBot",
   "user_agent_example": "Mozilla/5.0 (compatible; SeznamBot/4.0; +http://napoveda.seznam.cz/en/seznambot-intro/)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Seznam's crawler — the dominant search engine in the Czech Republic and one of the few national engines with its own index.",
   "cost_of_blocking": "Removal from Seznam. Also removes you from its IndexNow endpoint's usefulness.",
   "operator_docs": "https://napoveda.seznam.cz/en/seznamzbozi/subject-matter-crawler/",
   "html_url": "https://www.pathwren.workers.dev/crawler/seznambot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/seznambot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "storebot-google",
   "name": "Storebot-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Storebot-Google",
   "user_agent_substring": "Storebot-Google",
   "user_agent_example": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Safari/537.36 (compatible; Storebot-Google/1.0; +https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "Checks shopping and checkout flows for Google's shopping surfaces.",
   "cost_of_blocking": "Product listings may lose shopping-specific enrichment. Irrelevant to non-commerce sites.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/storebot-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/storebot-google.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "timpibot",
   "name": "Timpibot",
   "operator": "Timpi",
   "operator_slug": "timpi",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Timpibot",
   "user_agent_substring": "Timpibot",
   "user_agent_example": "Mozilla/5.0 (compatible; Timpibot/0.1; +https://timpi.io)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A distributed crawler building an independent search index outside the Google/Bing duopoly.",
   "cost_of_blocking": "Absence from a small independent index.",
   "operator_docs": "https://timpi.io/",
   "html_url": "https://www.pathwren.workers.dev/crawler/timpibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/timpibot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexblogs",
   "name": "YandexBlogs",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexBlogs",
   "user_agent_substring": "YandexBlogs",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexBlogs/0.99; robot; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Yandex's blog-search robot; it indexes post comments as well as posts.",
   "cost_of_blocking": "Comment threads and blog posts stop being findable through Yandex blog search.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexblogs.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexblogs.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexbot",
   "name": "YandexBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexBot",
   "user_agent_substring": "YandexBot",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexBot/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Yandex's search crawler, which also feeds Alice and Yandex's generative answers.",
   "cost_of_blocking": "Removal from Yandex Search. Verify with reverse DNS to a yandex.ru, yandex.net or yandex.com host — YandexBot is among the most-spoofed user-agents there is.",
   "operator_docs": "https://yandex.com/support/webmaster/robot-workings/check-yandex-robots.html",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexcombot",
   "name": "YandexComBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexComBot",
   "user_agent_substring": "YandexComBot",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexComBot/3.0; +http://ya.cc/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes content for Yandex search in languages other than Russian. Yandex documents that it can index content when there is no explicit robot-specific restriction — a * group is not one.",
   "cost_of_blocking": "You leave Yandex's non-Russian index. A rule naming this token is the only one that works on it.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexcombot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexcombot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexfavicons",
   "name": "YandexFavicons",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexFavicons",
   "user_agent_substring": "YandexFavicons",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexFavicons/1.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Downloads your favicon so Yandex can show it beside your result. Documented as not taking the general robots.txt rules into account.",
   "cost_of_blocking": "Your results in Yandex lose their icon. Cosmetic, and a * rule will not achieve it anyway.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexfavicons.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexfavicons.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandeximages",
   "name": "YandexImages",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexImages",
   "user_agent_substring": "YandexImages",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexImages/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes images for Yandex Images. Yandex's robot table marks it as taking the general robots.txt rules into account.",
   "cost_of_blocking": "Your images leave Yandex Images, which is a large share of image search in Russian-speaking markets.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandeximages.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandeximages.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexmarket",
   "name": "YandexMarket",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexMarket",
   "user_agent_substring": "YandexMarket",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexMarket/1.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The robot behind Yandex Market, Yandex's shopping comparison service. Version 1.0 is documented as obeying the general rules; version 2.0 is documented as not.",
   "cost_of_blocking": "Your products stop being listed and priced in Yandex Market. For a retailer in that market this is a revenue block, not a bandwidth one.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexmarket.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexmarket.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexmedia",
   "name": "YandexMedia",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexMedia",
   "user_agent_substring": "YandexMedia",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexMedia/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes multimedia data for Yandex. Takes the general robots.txt rules into account.",
   "cost_of_blocking": "Your multimedia content stops appearing in Yandex's media surfaces.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexmedia.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexmedia.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexmobilebot",
   "name": "YandexMobileBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexMobileBot",
   "user_agent_substring": "YandexMobileBot",
   "user_agent_example": "Mozilla/5.0 (iPhone; CPU iPhone OS 8_1 like Mac OS X) AppleWebKit/600.1.4 (KHTML, like Gecko) Version/8.0 Mobile/12B411 Safari/600.1.4 (compatible; YandexMobileBot/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Decides whether a page's layout is suitable for mobile devices. Yandex's table marks it as NOT taking the general robots.txt rules into account, so a * group does not stop it — a group named YandexMobileBot does.",
   "cost_of_blocking": "Yandex loses its mobile-friendliness signal for your pages, which affects how they are ranked and rendered on phones.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexmobilebot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexmobilebot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexrenderresourcesbot",
   "name": "YandexRenderResourcesBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexRenderResourcesBot",
   "user_agent_substring": "YandexRenderResourcesBot",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexRenderResourcesBot/1.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Loads the CSS, JavaScript and images Yandex needs to render a page. Yandex documents the exact rule: it ignores robots.txt for a resource when the HTML page using it is allowed, and does not fetch the resource when that page is disallowed.",
   "cost_of_blocking": "Yandex renders your pages without their stylesheets or scripts and ranks what it sees. This is the classic accidental self-inflicted ranking loss.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexrenderresourcesbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexrenderresourcesbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexvideo",
   "name": "YandexVideo",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "YandexVideo",
   "user_agent_substring": "YandexVideo",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexVideo/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Indexes video for Yandex video search. Obeys the general robots.txt rules per Yandex's own table.",
   "cost_of_blocking": "Removal from Yandex video search. Note that a second robot, YandexVideoParser, does the same job and is documented as NOT taking the general rules into account.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexvideo.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexvideo.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yeti",
   "name": "Yeti",
   "operator": "Naver",
   "operator_slug": "naver",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Yeti",
   "user_agent_substring": "Yeti",
   "user_agent_example": "Mozilla/5.0 (compatible; Yeti/1.1; +https://naver.me/spd)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Naver's crawler. Naver is South Korea's largest search portal and runs its own index and its own generative answers.",
   "cost_of_blocking": "Removal from Naver, which is most of Korean search.",
   "operator_docs": "https://searchadvisor.naver.com/guide/seo-basic-crawl",
   "html_url": "https://www.pathwren.workers.dev/crawler/yeti.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yeti.json",
   "last_reviewed": "2026-09-03"
  }
 ]
}