{
 "category": "tool",
 "label": "Tools and frameworks",
 "description": "Not operators: crawling software anyone can run. The party behind the request is unknown, so treat them as a rate-limit question rather than a consent question.",
 "count": 22,
 "crawlers": [
  {
   "slug": "adsbot-google",
   "name": "AdsBot-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "AdsBot-Google",
   "user_agent_substring": "AdsBot-Google",
   "user_agent_example": "AdsBot-Google (+http://www.google.com/adsbot.html)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "Checks the quality of desktop landing pages for Google Ads. Google documents that it ignores the robots.txt * group with the ad publisher's permission, and obeys a group named for its own token.",
   "cost_of_blocking": "Google Ads cannot score your landing pages, which lowers Ad Rank on the ads pointing at them. If you do not buy ads, blocking it costs nothing but bandwidth savings.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/adsbot-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/adsbot-google.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "adsbot-google-mobile",
   "name": "AdsBot-Google-Mobile",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "AdsBot-Google-Mobile",
   "user_agent_substring": "AdsBot-Google-Mobile",
   "user_agent_example": "Mozilla/5.0 (Linux; Android 5.0; SM-G920A) AppleWebKit (KHTML, like Gecko) Chrome Mobile Safari (compatible; AdsBot-Google-Mobile; +http://www.google.com/mobile/adsbot.html)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "The mobile-web landing page checker for Google Ads. Same rules as AdsBot-Google: the * group does not apply to it, its own token does.",
   "cost_of_blocking": "Mobile ad landing pages go unscored and the ads pointing at them rank worse. No effect on organic search.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/adsbot-google-mobile.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/adsbot-google-mobile.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "adsbot-google-mobile-apps",
   "name": "AdsBot-Google-Mobile-Apps",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "AdsBot-Google-Mobile-Apps",
   "user_agent_substring": "AdsBot-Google-Mobile-Apps",
   "user_agent_example": "AdsBot-Google-Mobile-Apps",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "Checks Android app landing pages for Google Ads. It obeys a group named for its own token and, per Google, follows the AdsBot-Google rules otherwise.",
   "cost_of_blocking": "App-install ad landing pages go unscored. Nothing organic changes.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/adsbot-google-mobile-apps.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/adsbot-google-mobile-apps.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "apis-google",
   "name": "APIs-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "APIs-Google",
   "user_agent_substring": "APIs-Google",
   "user_agent_example": "APIs-Google (+https://developers.google.com/webmasters/APIs-Google.html)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "Delivers push notifications for Google APIs to a webhook you registered. It is a special-case crawler: it ignores the robots.txt * group, because the fetch is a delivery to an address you asked it to deliver to.",
   "cost_of_blocking": "Google API push notifications stop arriving at your endpoint. This only affects services you set up yourself; there is no search or AI consequence.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/apis-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/apis-google.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "crawl4ai",
   "name": "Crawl4AI",
   "operator": "Crawl4AI project",
   "operator_slug": "crawl4ai",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Crawl4AI",
   "user_agent_substring": "Crawl4AI",
   "user_agent_example": "Crawl4AI",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "An open-source LLM-oriented crawler and scraper library, run by whoever installs it. Like Scrapy, the default user-agent identifies the software and says nothing about who is behind the request.",
   "cost_of_blocking": "You block a library, not an operator: the rule catches a researcher and a bulk scraper equally, and anyone who edits one config line is not caught at all.",
   "operator_docs": "https://github.com/unclecode/crawl4ai",
   "html_url": "https://www.pathwren.workers.dev/crawler/crawl4ai.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/crawl4ai.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "crawlspace",
   "name": "Crawlspace",
   "operator": "Crawlspace",
   "operator_slug": "crawlspace",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Crawlspace",
   "user_agent_substring": "Crawlspace",
   "user_agent_example": "Crawlspace",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A crawling platform: customers run their own crawls on it to feed agents, RAG pipelines and structured-data workflows. Like Firecrawl, the party behind any given request is the customer, not the platform.",
   "cost_of_blocking": "Whatever any Crawlspace customer was building over your pages stops working. Volume and intent vary per customer, so this is a rate-limit decision more than a consent one.",
   "operator_docs": "https://crawlspace.dev",
   "html_url": "https://www.pathwren.workers.dev/crawler/crawlspace.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/crawlspace.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "feedfetcher-google",
   "name": "FeedFetcher-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "FeedFetcher-Google",
   "user_agent_substring": "FeedFetcher-Google",
   "user_agent_example": "FeedFetcher-Google; (+http://www.google.com/feedfetcher.html)",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered.json",
   "ipv4_prefix_count": 529,
   "ipv6_prefix_count": 529,
   "what_it_is": "Crawls RSS and Atom feeds for Google News and WebSub. It is a user-triggered fetcher, and Google documents that those generally ignore robots.txt because a person asked for the fetch. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "cost_of_blocking": "Feed-driven Google products stop seeing your updates. A robots.txt rule will not stop it — block by user-agent at the edge if you mean it.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-user-triggered-fetchers",
   "html_url": "https://www.pathwren.workers.dev/crawler/feedfetcher-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/feedfetcher-google.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "firecrawlagent",
   "name": "FirecrawlAgent",
   "operator": "Firecrawl",
   "operator_slug": "firecrawl",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "FirecrawlAgent",
   "user_agent_substring": "FirecrawlAgent",
   "user_agent_example": "Mozilla/5.0 (compatible; FirecrawlAgent/1.0; +https://firecrawl.dev)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A hosted scrape-to-markdown service that LLM applications call to read pages. The requester is whoever is building on it, not Firecrawl itself, so volume and intent vary wildly.",
   "cost_of_blocking": "Applications built on Firecrawl cannot read your pages. This is increasingly how agents fetch the web, so it is a bigger block than its name suggests.",
   "operator_docs": "https://docs.firecrawl.dev/",
   "html_url": "https://www.pathwren.workers.dev/crawler/firecrawlagent.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/firecrawlagent.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-cws",
   "name": "Google-CWS",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Google-CWS",
   "user_agent_substring": "Google-CWS",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-CWS)",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered.json",
   "ipv4_prefix_count": 529,
   "ipv6_prefix_count": 529,
   "what_it_is": "The Chrome Web Store fetcher. It requests the URLs a developer put in the metadata of a Chrome extension or theme. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "cost_of_blocking": "Chrome Web Store listings that point at your pages cannot fetch them. Relevant only if you publish extensions.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-user-triggered-fetchers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-cws.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-cws.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-inspectiontool",
   "name": "Google-InspectionTool",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Google-InspectionTool",
   "user_agent_substring": "Google-InspectionTool",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-InspectionTool/1.0;)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "The fetcher behind Search Console's URL Inspection and the Rich Results Test. It runs when a site owner clicks a button.",
   "cost_of_blocking": "Your own Search Console live tests stop working. Blocking this only hurts you.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-inspectiontool.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-inspectiontool.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-safety",
   "name": "Google-Safety",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Google-Safety",
   "user_agent_substring": "Google-Safety",
   "user_agent_example": "Google-Safety",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Google's abuse-investigation fetcher: malware review, phishing reports and similar. Google documents that it ignores robots.txt entirely, and a robots.txt rule for it does nothing.",
   "cost_of_blocking": "Nothing you can control. The rule is ignored by design; listing the token is documentation, not enforcement.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-safety.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-safety.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "google-site-verification",
   "name": "Google-Site-Verification",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Google-Site-Verification",
   "user_agent_substring": "Google-Site-Verification",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-Site-Verification/1.0)",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered.json",
   "ipv4_prefix_count": 529,
   "ipv6_prefix_count": 529,
   "what_it_is": "Fetches the token file or meta tag that proves you own a site, when you click verify in Search Console. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "cost_of_blocking": "Your own Search Console verification fails. Blocking this only ever hurts the person doing the blocking.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-user-triggered-fetchers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-site-verification.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-site-verification.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "googleproducer",
   "name": "GoogleProducer",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "GoogleProducer",
   "user_agent_substring": "GoogleProducer",
   "user_agent_example": "GoogleProducer; (+https://developers.google.com/search/docs/crawling-indexing/google-producer)",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered.json",
   "ipv4_prefix_count": 529,
   "ipv6_prefix_count": 529,
   "what_it_is": "Google Publisher Center: fetches the feeds a publisher explicitly supplied for Google News landing pages. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "cost_of_blocking": "Your own Google News landing pages stop updating. Only publishers who configured Publisher Center are affected.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-user-triggered-fetchers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googleproducer.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googleproducer.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "lightpanda",
   "name": "Lightpanda",
   "operator": "Lightpanda",
   "operator_slug": "lightpanda",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Lightpanda",
   "user_agent_substring": "Lightpanda",
   "user_agent_example": "Lightpanda",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "A purpose-built headless browser for AI and automation — a runtime, not an operator. Whether robots.txt is honoured is left to whoever runs it, which is what its maintainers say themselves.",
   "cost_of_blocking": "You block a browser, not a company: the same rule stops a scraper and a legitimate automation a customer of yours is running.",
   "operator_docs": "https://lightpanda.io/",
   "html_url": "https://www.pathwren.workers.dev/crawler/lightpanda.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/lightpanda.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "mediapartners-google",
   "name": "Mediapartners-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Mediapartners-Google",
   "user_agent_substring": "Mediapartners-Google",
   "user_agent_example": "Mediapartners-Google",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 136,
   "ipv6_prefix_count": 136,
   "what_it_is": "The AdSense crawler. It reads a page so AdSense can choose relevant ads for it, and it is a special-case crawler that ignores the robots.txt * group.",
   "cost_of_blocking": "Pages it cannot read get generic, lower-value AdSense ads or none at all. This is the one block on this list that costs you money directly if you run AdSense.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/mediapartners-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/mediapartners-google.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "scrapy",
   "name": "Scrapy",
   "operator": "Scrapy project",
   "operator_slug": "scrapy",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Scrapy",
   "user_agent_substring": "Scrapy",
   "user_agent_example": "Scrapy/2.11.0 (+https://scrapy.org)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Not an operator: the default user-agent of the most common Python crawling framework. Anyone can be behind it. Modern Scrapy obeys robots.txt by default, which is why the default UA is still worth a rule.",
   "cost_of_blocking": "You block a very large tail of unattributed one-off crawlers, and also every well-behaved researcher who did not change the default.",
   "operator_docs": "https://scrapy.org/",
   "html_url": "https://www.pathwren.workers.dev/crawler/scrapy.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/scrapy.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "screaming-frog-seo-spider",
   "name": "Screaming Frog SEO Spider",
   "operator": "Screaming Frog",
   "operator_slug": "screamingfrog",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Screaming Frog SEO Spider",
   "user_agent_substring": "Screaming Frog SEO Spider",
   "user_agent_example": "Screaming Frog SEO Spider/21.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Not an operator: desktop crawling software that anybody can point at any site. The default user-agent identifies the tool, not who is running it, and the operator of the moment is whoever pressed start.",
   "cost_of_blocking": "You block a consultant auditing your own site as often as you block a stranger. Treat it as a rate-limit question, not a consent one — and note that the user-agent is configurable, so a block is advisory.",
   "operator_docs": "https://www.screamingfrog.co.uk/seo-spider/user-agent/",
   "html_url": "https://www.pathwren.workers.dev/crawler/screaming-frog-seo-spider.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/screaming-frog-seo-spider.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "wpbot",
   "name": "wpbot",
   "operator": "QuantumCloud",
   "operator_slug": "quantumcloud",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "wpbot",
   "user_agent_substring": "wpbot",
   "user_agent_example": "wpbot",
   "respects_robots_txt": "undocumented",
   "respects_robots_txt_label": "operator publishes no robots.txt statement",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Supports the AI Chatbot for WordPress plugin: it reads pages so the plugin can answer from a site's own content. The operator provides an opt-out through a form rather than through robots.txt.",
   "cost_of_blocking": "A WordPress site running that plugin loses its own content as an answer source. Only relevant where the plugin is installed.",
   "operator_docs": "https://www.quantumcloud.com",
   "html_url": "https://www.pathwren.workers.dev/crawler/wpbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/wpbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexdirect",
   "name": "YandexDirect",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "YandexDirect",
   "user_agent_substring": "YandexDirect",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexDirect/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Reads the content of Yandex Advertising Network partner pages to work out their topic so relevant ads can be matched. Documented as not taking the general robots.txt rules into account.",
   "cost_of_blocking": "Ads on your pages become less relevant and earn less. Relevant only if you monetise with Yandex's network.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexdirect.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexdirect.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexmetrika",
   "name": "YandexMetrika",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "YandexMetrika",
   "user_agent_substring": "YandexMetrika",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexMetrika/2.0; +http://yandex.com/bots)",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Yandex Metrica's own fetcher. Two of its versions — the 2.0 yabs01 availability checker and the 4.0 CSS cache for Webvisor — are documented in Yandex's table as not using robots.txt at all.",
   "cost_of_blocking": "Nothing you can enforce through robots.txt. If you run Metrica, its session replay loses your stylesheets and renders your pages wrong.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexmetrika.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexmetrika.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexscreenshotbot",
   "name": "YandexScreenshotBot",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "YandexScreenshotBot",
   "user_agent_substring": "YandexScreenshotBot",
   "user_agent_example": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Safari/537.36 (compatible; YandexScreenshotBot/3.0; +http://yandex.com/bots)",
   "respects_robots_txt": "own-token-only",
   "respects_robots_txt_label": "ignores the * group; obeys rules named for its own token",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Takes a screenshot of a page. Documented as not taking the general robots.txt rules into account.",
   "cost_of_blocking": "Yandex surfaces that show a page thumbnail show nothing for you.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexscreenshotbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexscreenshotbot.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "yandexwebmaster",
   "name": "YandexWebmaster",
   "operator": "Yandex",
   "operator_slug": "yandex",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "YandexWebmaster",
   "user_agent_substring": "YandexWebmaster",
   "user_agent_example": "Mozilla/5.0 (compatible; YandexWebmaster/2.0; +http://yandex.com/bots)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "reverse-dns",
   "verification_label": "reverse DNS",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The fetcher behind Yandex Webmaster, the console a site owner uses to inspect their own site.",
   "cost_of_blocking": "Your own Yandex Webmaster checks stop working. Blocking this only hurts you.",
   "operator_docs": "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots",
   "html_url": "https://www.pathwren.workers.dev/crawler/yandexwebmaster.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/yandexwebmaster.json",
   "last_reviewed": "2026-09-03"
  }
 ]
}