{
 "swagger": "2.0",
 "info": {
  "title": "AI Crawler Index",
  "version": "2026-09-02",
  "description": "Every AI crawler on the web, what it is for, what blocking it costs you, and the IP ranges its operator publishes — as JSON, CSV, robots.txt and regex.\n\nA read-only, static, keyless index of 150 web crawlers operated by 74 companies and projects: what each one is for, the exact robots.txt token and user-agent substring, whether the operator says it obeys robots.txt, how to verify it is genuine, and — the part nobody else publishes — what you lose by blocking it.\n\nEvery path below is a static file. There is no key, no rate limit and no state; `Access-Control-Allow-Origin: *` is set, so it is callable from a browser. Data is CC0.\n\nThe IP-range mirrors are refreshed from 15 operator-published endpoints; /status.json reports when each was last fetched and whether it changed.\n\nMechanically converted from https://www.pathwren.workers.dev/openapi.json (OpenAPI 3.1) by the same build, in the same run, from the same object — this is not a second hand-maintained description and cannot drift from the first. Every operation is a keyless GET of a static file, which 2.0 describes as completely as 3.1 does. One 3.1 construct has no 2.0 equivalent and is rendered with the usual extension instead: Crawler.published_ip_ranges_url, GET /changes.json: kept the first of 3 examples is `type: string` with `x-nullable: true` (2.0 has no null type). It is null when the operator publishes no IP ranges. The 3.1 original is canonical and is what /apis.json, /.well-known/api-catalog and the MCP server point at.",
  "license": {
   "name": "CC0-1.0",
   "url": "https://creativecommons.org/publicdomain/zero/1.0/"
  },
  "termsOfService": "https://www.pathwren.workers.dev/terms.html",
  "contact": {
   "url": "https://www.pathwren.workers.dev/about.html"
  }
 },
 "host": "www.pathwren.workers.dev",
 "basePath": "/",
 "schemes": [
  "https"
 ],
 "produces": [
  "application/json"
 ],
 "paths": {
  "/data/agents.json": {
   "get": {
    "tags": [
     "bulk"
    ],
    "summary": "Every crawler record, plus categories and an endpoint map",
    "operationId": "listCrawlers",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "150 records",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "object",
       "properties": {
        "count": {
         "type": "integer"
        },
        "generated_at": {
         "type": "string",
         "format": "date-time"
        },
        "crawlers": {
         "type": "array",
         "items": {
          "$ref": "#/definitions/Crawler"
         }
        }
       }
      },
      "examples": {
       "application/json": {
        "count": 150,
        "generated_at": "2026-09-02T03:40:24+00:00",
        "crawlers": [
         {
          "slug": "oai-searchbot",
          "name": "OAI-SearchBot",
          "operator": "OpenAI",
          "operator_slug": "openai",
          "category": "ai-search",
          "category_label": "AI search crawlers",
          "robots_token": "OAI-SearchBot",
          "user_agent_substring": "OAI-SearchBot",
          "user_agent_example": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-SearchBot/1.0; +https://openai.com/searchbot",
          "respects_robots_txt": "documented",
          "respects_robots_txt_label": "obeys robots.txt (documented)",
          "verification_method": "published-ranges",
          "verification_label": "published IP ranges",
          "published_ip_ranges_url": "https://openai.com/searchbot.json",
          "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/openai-searchbot.json",
          "ipv4_prefix_count": 35,
          "ipv6_prefix_count": 0,
          "what_it_is": "Builds the index ChatGPT search answers from. Content it collects is used for retrieval and citation, not for model training.",
          "cost_of_blocking": "High. Blocking this removes you from ChatGPT search results and from the source links ChatGPT shows. This is the single most expensive block on this list for anyone who wants to be cited by an assistant.",
          "operator_docs": "https://platform.openai.com/docs/bots",
          "html_url": "https://www.pathwren.workers.dev/crawler/oai-searchbot.html",
          "json_url": "https://www.pathwren.workers.dev/crawler/oai-searchbot.json",
          "last_reviewed": "2026-09-02"
         }
        ]
       }
      }
     }
    }
   }
  },
  "/crawler/{slug}.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "One crawler record by slug",
    "description": "The same records as /data/agents.json, one file each, so a client can fetch exactly the one it needs.",
    "operationId": "getCrawler",
    "parameters": [
     {
      "name": "slug",
      "in": "path",
      "required": true,
      "description": "Crawler slug, from /data/agents.json",
      "type": "string",
      "enum": [
       "gptbot",
       "oai-searchbot",
       "chatgpt-user",
       "claudebot",
       "claude-searchbot",
       "claude-user",
       "anthropic-ai",
       "claude-web",
       "google-extended",
       "googlebot",
       "googleother",
       "google-cloudvertexbot",
       "google-inspectiontool",
       "googlebot-image",
       "googlebot-news",
       "storebot-google",
       "bingbot",
       "applebot",
       "applebot-extended",
       "perplexitybot",
       "perplexity-user",
       "ccbot",
       "bytespider",
       "tiktokspider",
       "meta-externalagent",
       "meta-externalfetcher",
       "facebookexternalhit",
       "facebookbot",
       "amazonbot",
       "duckassistbot",
       "duckduckbot",
       "ai2bot",
       "ai2bot-dolma",
       "cohere-ai",
       "cohere-training-data-crawler",
       "mistralai-user",
       "youbot",
       "diffbot",
       "omgilibot",
       "omgili",
       "webzio-extended",
       "imagesiftbot",
       "timpibot",
       "semrushbot",
       "semrushbot-ocob",
       "ahrefsbot",
       "archive-org-bot",
       "ia-archiver",
       "yandexbot",
       "baiduspider",
       "seznambot",
       "yeti",
       "petalbot",
       "firecrawlagent",
       "scrapy",
       "img2dataset",
       "googlebot-video",
       "googleother-image",
       "googleother-video",
       "apis-google",
       "adsbot-google",
       "adsbot-google-mobile",
       "adsbot-google-mobile-apps",
       "mediapartners-google",
       "google-safety",
       "feedfetcher-google",
       "google-read-aloud",
       "google-site-verification",
       "google-cws",
       "google-pinpoint",
       "googleproducer",
       "googlemessages",
       "google-gemininotebook",
       "google-agent",
       "yandeximages",
       "yandexvideo",
       "yandexmedia",
       "yandexblogs",
       "yandexmarket",
       "yandexwebmaster",
       "yandexmobilebot",
       "yandexfavicons",
       "yandexcalendar",
       "yandexdirect",
       "yandexmetrika",
       "yandexrenderresourcesbot",
       "yandexscreenshotbot",
       "yandexadditional",
       "yandexadditionalbot",
       "yandexcombot",
       "siteauditbot",
       "semrushbot-ba",
       "semrushbot-si",
       "semrushbot-swa",
       "splitsignalbot",
       "semrushbot-ft",
       "semrushbot-esi",
       "ahrefssiteaudit",
       "mj12bot",
       "dotbot",
       "rogerbot",
       "dataforseobot",
       "serpstatbot",
       "barkrowler",
       "screaming-frog-seo-spider",
       "seokicks",
       "mojeekbot",
       "kagibot",
       "qwantbot",
       "qwantbot-news",
       "slackbot-linkexpanding",
       "slackbot",
       "pinterestbot",
       "bedrockbot",
       "cloudflare-autorag",
       "exasearchbot",
       "shapbot",
       "terracotta",
       "crawlspace",
       "panscient",
       "sbintuitionsbot",
       "icc-crawler",
       "cotoyogi",
       "isscyberriskcrawler",
       "sidetrade-indexer-bot",
       "yak",
       "atlassian-bot",
       "klaviyoaibot",
       "quillbot",
       "phindbot",
       "andibot",
       "anomura",
       "aiwebindex",
       "factset-spyderbot",
       "poseidon-research-crawler",
       "qualifiedbot",
       "reflectionbot",
       "thinkbot",
       "aihitbot",
       "linguee-bot",
       "lightpanda",
       "laiondownloader",
       "velenpublicwebcrawler",
       "awariosmartbot",
       "awariorssbot",
       "echoboxbot",
       "meta-webindexer",
       "chatgpt-agent",
       "wpbot",
       "crawl4ai"
      ]
     }
    ],
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      },
      "examples": {
       "application/json": {
        "slug": "gptbot",
        "name": "GPTBot",
        "operator": "OpenAI",
        "operator_slug": "openai",
        "category": "ai-training",
        "category_label": "AI training crawlers",
        "robots_token": "GPTBot",
        "user_agent_substring": "GPTBot",
        "user_agent_example": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.2; +https://openai.com/gptbot",
        "respects_robots_txt": "documented",
        "respects_robots_txt_label": "obeys robots.txt (documented)",
        "verification_method": "published-ranges",
        "verification_label": "published IP ranges",
        "published_ip_ranges_url": "https://openai.com/gptbot.json",
        "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/openai-gptbot.json",
        "ipv4_prefix_count": 21,
        "ipv6_prefix_count": 0,
        "what_it_is": "OpenAI's bulk crawler. Pages it fetches may be used to train future OpenAI foundation models. It is not the bot that puts you in ChatGPT's search results, and blocking it does not remove you from them.",
        "cost_of_blocking": "Your content is excluded from training data for future OpenAI models. No effect on ChatGPT search visibility, on citations, or on links a user pastes into ChatGPT.",
        "operator_docs": "https://platform.openai.com/docs/bots",
        "html_url": "https://www.pathwren.workers.dev/crawler/gptbot.html",
        "json_url": "https://www.pathwren.workers.dev/crawler/gptbot.json",
        "last_reviewed": "2026-09-02"
       }
      }
     },
     "404": {
      "description": "No such crawler"
     }
    }
   }
  },
  "/data/agents.csv": {
   "get": {
    "tags": [
     "bulk"
    ],
    "summary": "The same table as CSV",
    "operationId": "listCrawlersCsv",
    "produces": [
     "text/csv"
    ],
    "responses": {
     "200": {
      "description": "CSV with a header row",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "string"
      }
     }
    }
   }
  },
  "/data/ua-regex.json": {
   "get": {
    "tags": [
     "bulk"
    ],
    "summary": "Pre-escaped user-agent regexes, whole-list and per category",
    "operationId": "getUserAgentRegex",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Regex alternations",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "object",
       "properties": {
        "all": {
         "type": "string"
        },
        "ai_only": {
         "type": "string"
        },
        "by_category": {
         "type": "object",
         "additionalProperties": {
          "type": "string"
         }
        }
       }
      }
     }
    }
   }
  },
  "/data/ip-sources.json": {
   "get": {
    "tags": [
     "ip-ranges"
    ],
    "summary": "Which operators publish verifiable IP ranges, and where",
    "operationId": "listIpSources",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Source list",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     }
    }
   }
  },
  "/ip-ranges/all.json": {
   "get": {
    "tags": [
     "ip-ranges"
    ],
    "summary": "Union of every operator-published prefix, grouped by source",
    "operationId": "getAllIpRanges",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "1984 IPv4 and 1062 IPv6 prefixes",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     }
    }
   }
  },
  "/ip-ranges/all.txt": {
   "get": {
    "tags": [
     "ip-ranges"
    ],
    "summary": "The same prefixes, one CIDR per line",
    "operationId": "getAllIpRangesText",
    "produces": [
     "text/plain"
    ],
    "responses": {
     "200": {
      "description": "Plain text, # comments",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "string"
      }
     }
    }
   }
  },
  "/ip-ranges/{source}.json": {
   "get": {
    "tags": [
     "ip-ranges"
    ],
    "summary": "One operator's published prefix list, normalised",
    "operationId": "getIpRangeSource",
    "parameters": [
     {
      "name": "source",
      "in": "path",
      "required": true,
      "description": "Source slug, from /data/ip-sources.json",
      "type": "string",
      "enum": [
       "openai-gptbot",
       "openai-searchbot",
       "openai-chatgpt-user",
       "google-googlebot",
       "google-special",
       "google-user-triggered",
       "google-user-triggered-google",
       "bing-bingbot",
       "apple-applebot",
       "duckduckgo-duckduckbot",
       "perplexity-bot",
       "perplexity-user",
       "google-user-triggered-agents",
       "commoncrawl-ccbot",
       "ahrefs-crawler"
      ]
     }
    ],
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Prefixes with provenance",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     },
     "404": {
      "description": "No such source"
     }
    }
   }
  },
  "/robots/{policy}.txt": {
   "get": {
    "tags": [
     "robots"
    ],
    "summary": "A ready-made robots.txt policy file",
    "operationId": "getRobotsPolicy",
    "parameters": [
     {
      "name": "policy",
      "in": "path",
      "required": true,
      "type": "string",
      "enum": [
       "allow-all",
       "block-ai-training",
       "block-all-ai",
       "block-datasets",
       "allow-ai-search-only",
       "block-seo-tools",
       "block-disputed",
       "maximum-ai-visibility"
      ]
     }
    ],
    "produces": [
     "text/plain"
    ],
    "responses": {
     "200": {
      "description": "robots.txt fragment, ready to append",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "string"
      }
     }
    }
   }
  },
  "/policy/{policy}.json": {
   "get": {
    "tags": [
     "robots"
    ],
    "summary": "A policy with its rationale and the crawlers it names",
    "operationId": "getPolicy",
    "parameters": [
     {
      "name": "policy",
      "in": "path",
      "required": true,
      "type": "string",
      "enum": [
       "allow-all",
       "block-ai-training",
       "block-all-ai",
       "block-datasets",
       "allow-ai-search-only",
       "block-seo-tools",
       "block-disputed",
       "maximum-ai-visibility"
      ]
     }
    ],
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Policy record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     }
    }
   }
  },
  "/documents.json": {
   "get": {
    "tags": [
     "status"
    ],
    "summary": "The document ledger: every URL here, its ETag and its last-modified",
    "description": "One request that tells you what to re-fetch. Every document this host publishes is listed with a strong ETag (sha-256 over the exact bytes served, first 32 hex characters) and the date those bytes last CHANGED. Compare against your copy, then GET only the paths that differ — with If-None-Match set to the etag from this file, so even a wrong guess costs a 304 and no body. HTML pages are listed with a null etag: live counters are rendered into them per request, so no validator for one could be honest.",
    "operationId": "getDocumentLedger",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Every published document with its validators",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "object",
       "properties": {
        "generated_at": {
         "type": "string",
         "format": "date-time"
        },
        "counts": {
         "type": "object"
        },
        "documents": {
         "type": "array",
         "items": {
          "type": "object",
          "properties": {
           "path": {
            "type": "string"
           },
           "kind": {
            "type": "string"
           },
           "type": {
            "type": "string"
           },
           "bytes": {
            "type": "integer"
           },
           "etag": {
            "type": "string"
           },
           "last_modified": {
            "type": "string"
           }
          }
         }
        }
       }
      }
     }
    }
   }
  },
  "/documents.txt": {
   "get": {
    "tags": [
     "status"
    ],
    "summary": "The same ledger as TSV: path, kind, bytes, etag, last-modified",
    "description": "For the client that is a shell. Comment lines start with '#'; every other line is five tab-separated fields. A '-' means no validator is published for that document.",
    "operationId": "getDocumentLedgerText",
    "produces": [
     "text/plain"
    ],
    "responses": {
     "200": {
      "description": "Tab-separated document ledger",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     }
    }
   }
  },
  "/status.json": {
   "get": {
    "tags": [
     "status"
    ],
    "summary": "Freshness and health of every upstream IP-range endpoint",
    "operationId": "getStatus",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "Per-source fetch state",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "type": "object",
       "properties": {
        "sources_ok": {
         "type": "integer"
        },
        "sources_failed": {
         "type": "integer"
        },
        "generated_at": {
         "type": "string",
         "format": "date-time"
        }
       }
      }
     }
    }
   }
  },
  "/feed.json": {
   "get": {
    "tags": [
     "status"
    ],
    "summary": "JSON Feed 1.1 of what changed",
    "operationId": "getFeed",
    "produces": [
     "application/feed+json"
    ],
    "responses": {
     "200": {
      "description": "JSON Feed",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      }
     }
    }
   }
  },
  "/changes.json": {
   "get": {
    "tags": [
     "changes"
    ],
    "summary": "What changed since your cursor",
    "description": "A since-cursor feed of every real change to this index: operator IP-range lists that gained or lost prefixes, upstreams that started or stopped answering, and crawler records added, edited or withdrawn.\n\nRead `cursor` from the response and send it back as `since` next time. The cursor is a monotonic integer that only advances when something actually changed, so an unchanged answer is proof of nothing new — not a coincidence of timing. An empty page is about 700 bytes; sending the response's ETag back as `If-None-Match` makes it a 304 with no body at all.\n\nThe 15 upstream endpoints are re-fetched every six hours, so polling faster than that returns the same cursor. Nothing is rate limited; the request is simply not worth your budget.\n\n`stale_sources` is repeated in EVERY response, including empty ones: when an upstream fails, this host keeps serving its last good prefixes rather than an empty list, and a client that polls daily should not have to have been listening at the exact minute it broke to find out.",
    "operationId": "getChanges",
    "parameters": [
     {
      "name": "since",
      "in": "query",
      "required": false,
      "description": "A cursor from a previous response, or an ISO-8601 timestamp. Omit it to read from the oldest retained event. An unparseable value is a 400, never a silently ignored parameter.",
      "type": "string"
     },
     {
      "name": "limit",
      "in": "query",
      "required": false,
      "description": "Events per page, 1–400. Default 100. Follow `next` when `has_more` is true.",
      "type": "integer",
      "default": 100,
      "minimum": 1,
      "maximum": 400
     }
    ],
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "A page of changes, oldest first",
      "headers": {
       "ETag": {
        "type": "string",
        "description": "Strong validator over these exact bytes. Send it as If-None-Match."
       },
       "Last-Modified": {
        "type": "string",
        "description": "When the newest fact in this answer was observed."
       },
       "X-Cursor": {
        "type": "string",
        "description": "The cursor to send as `since` next time; same as body.cursor."
       }
      },
      "schema": {
       "type": "object",
       "required": [
        "cursor",
        "head",
        "count",
        "changes"
       ],
       "properties": {
        "cursor": {
         "type": "integer",
         "description": "Send this back as `since`."
        },
        "head": {
         "type": "integer",
         "description": "Newest cursor that exists."
        },
        "up_to_date": {
         "type": "boolean"
        },
        "count": {
         "type": "integer"
        },
        "has_more": {
         "type": "boolean"
        },
        "next": {
         "type": "string",
         "format": "uri"
        },
        "cursor_expired": {
         "type": "boolean",
         "description": "Your cursor is older than the oldest retained event, so this page cannot be complete. Re-read the documents in `refetch` once and poll from `cursor`."
        },
        "stale_sources": {
         "type": "array",
         "description": "Upstreams that did not answer on the last refresh. Their last good prefixes are still served.",
         "items": {
          "type": "object"
         }
        },
        "changes": {
         "type": "array",
         "items": {
          "$ref": "#/definitions/Change"
         }
        }
       }
      }
     },
     "304": {
      "description": "Nothing has changed since the ETag or Last-Modified you sent. No body."
     },
     "400": {
      "description": "`since` was neither a cursor nor a timestamp. The body names the accepted forms."
     }
    }
   }
  },
  "/crawler/gptbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GPTBot",
    "description": "OpenAI's bulk crawler. Pages it fetches may be used to train future OpenAI foundation models. It is not the bot that puts you in ChatGPT's search results, and blocking it does not remove you from them.",
    "operationId": "crawler_gptbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/oai-searchbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for OAI-SearchBot",
    "description": "Builds the index ChatGPT search answers from. Content it collects is used for retrieval and citation, not for model training.",
    "operationId": "crawler_oai_searchbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/chatgpt-user.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ChatGPT-User",
    "description": "Fetches a single page at the moment a user or a ChatGPT agent asks for it — a pasted link, a browsing step, an Operator task. One human intent, one request. OpenAI states these fetches are not used for training.",
    "operationId": "crawler_chatgpt_user",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/claudebot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ClaudeBot",
    "description": "Anthropic's bulk crawler, gathering pages that may be used to train Claude models.",
    "operationId": "crawler_claudebot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/claude-searchbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Claude-SearchBot",
    "description": "Indexes pages so Claude's web search can find and cite them. Separate token from the training crawler, so search visibility and training consent are independent decisions.",
    "operationId": "crawler_claude_searchbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/claude-user.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Claude-User",
    "description": "Fetches a page because a Claude user asked Claude to read it, at that moment.",
    "operationId": "crawler_claude_user",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/anthropic-ai.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for anthropic-ai",
    "description": "A legacy robots.txt token from before Anthropic consolidated on ClaudeBot. It is still widely present in robots.txt files and costs nothing to keep, but it is a control token rather than a bot you will see in logs.",
    "operationId": "crawler_anthropic_ai",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/claude-web.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Claude-Web",
    "description": "An earlier Anthropic token for user-facing web access, superseded by Claude-User and Claude-SearchBot. Kept here because it appears in most published robots.txt templates.",
    "operationId": "crawler_claude_web",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-extended.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Extended",
    "description": "Not a crawler. A robots.txt token that tells Google whether pages Googlebot already fetched may be used to train and ground Gemini. You will never see it in an access log; disallowing it changes what Google does with content it fetched under a different name.",
    "operationId": "crawler_google_extended",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googlebot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Googlebot",
    "description": "The classic search crawler. It is also the crawler behind AI Overviews: Google does not run a separate bot for them, which is why the only AI opt-out is the Google-Extended token and not a Googlebot block.",
    "operationId": "crawler_googlebot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googleother.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GoogleOther",
    "description": "A generic fetcher used by Google product teams for one-off crawls and research, including data collection that does not belong to Search.",
    "operationId": "crawler_googleother",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-cloudvertexbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-CloudVertexBot",
    "description": "Crawls a site on behalf of a Vertex AI Agent Builder customer who is building an agent over that site. It only visits sites the customer has asked it to.",
    "operationId": "crawler_google_cloudvertexbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-inspectiontool.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-InspectionTool",
    "description": "The fetcher behind Search Console's URL Inspection and the Rich Results Test. It runs when a site owner clicks a button.",
    "operationId": "crawler_google_inspectiontool",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googlebot-image.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Googlebot-Image",
    "description": "Image indexing for Google Images. A separate token so you can leave images out of search without leaving search.",
    "operationId": "crawler_googlebot_image",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googlebot-news.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Googlebot-News",
    "description": "A robots.txt token controlling inclusion in Google News. It does not have its own user-agent string; the fetch arrives as Googlebot.",
    "operationId": "crawler_googlebot_news",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/storebot-google.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Storebot-Google",
    "description": "Checks shopping and checkout flows for Google's shopping surfaces.",
    "operationId": "crawler_storebot_google",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/bingbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for bingbot",
    "description": "Bing's only crawler, and therefore also the crawler behind Microsoft Copilot's grounding. Microsoft's documented way to keep search indexing while refusing generative reuse is the nocache / noarchive robots meta directive, not a separate user-agent.",
    "operationId": "crawler_bingbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/applebot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Applebot",
    "description": "Powers Siri, Spotlight and Safari suggestions. Blocking it is a search decision, not an AI decision — the AI decision has its own token.",
    "operationId": "crawler_applebot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/applebot-extended.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Applebot-Extended",
    "description": "Apple's counterpart to Google-Extended: a robots.txt token that withdraws consent for Apple Intelligence and Apple foundation-model training, without touching Applebot's search crawl.",
    "operationId": "crawler_applebot_extended",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/perplexitybot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for PerplexityBot",
    "description": "Builds Perplexity's search index. Perplexity is citation-heavy by product design, so inclusion here converts to referral traffic more directly than most AI surfaces.",
    "operationId": "crawler_perplexitybot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/perplexity-user.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Perplexity-User",
    "description": "Fetches a page because a Perplexity user asked for it. Perplexity documents that this fetch is user-initiated and is therefore not governed by robots.txt — a robots rule will not stop it, by stated policy.",
    "operationId": "crawler_perplexity_user",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ccbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for CCBot",
    "description": "Common Crawl's corpus builder. It trains nothing itself, but its archive is an input to most open and many closed LLM training sets, which makes it the highest-leverage single entry on this list.",
    "operationId": "crawler_ccbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/bytespider.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Bytespider",
    "description": "ByteDance's crawler, associated with training data collection for Doubao and related models. Repeatedly reported by CDNs and site operators as the highest-volume AI crawler on the web and as inconsistent about robots.txt.",
    "operationId": "crawler_bytespider",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/tiktokspider.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for TikTokSpider",
    "description": "A second ByteDance crawler identifying with TikTok, collecting page content for the same family of models.",
    "operationId": "crawler_tiktokspider",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/meta-externalagent.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for meta-externalagent",
    "description": "Meta's AI crawler, gathering training data for Llama and Meta AI. It replaced the older FacebookBot name for this purpose.",
    "operationId": "crawler_meta_externalagent",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/meta-externalfetcher.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for meta-externalfetcher",
    "description": "Fetches a page when a Meta AI user asks about a specific link.",
    "operationId": "crawler_meta_externalfetcher",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/facebookexternalhit.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for facebookexternalhit",
    "description": "The link unfurler: it reads your Open Graph tags when somebody shares your URL on a Meta property.",
    "operationId": "crawler_facebookexternalhit",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/facebookbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for FacebookBot",
    "description": "Meta's older speech- and language-corpus crawler, largely superseded by meta-externalagent but still listed as a valid robots token.",
    "operationId": "crawler_facebookbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/amazonbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Amazonbot",
    "description": "Amazon's crawler, feeding Alexa's ability to answer questions from the web and Amazon's own search and assistant products.",
    "operationId": "crawler_amazonbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/duckassistbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for DuckAssistBot",
    "description": "Fetches pages so DuckAssist can generate and cite answers inside DuckDuckGo.",
    "operationId": "crawler_duckassistbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/duckduckbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for DuckDuckBot",
    "description": "DuckDuckGo's own crawler. Note that the bulk of DuckDuckGo's web results come from Bing, so blocking bingbot removes you from DuckDuckGo whether or not you allow this one.",
    "operationId": "crawler_duckduckbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ai2bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AI2Bot",
    "description": "The Allen Institute's crawler, gathering pages for open research corpora such as Dolma that underpin fully open models like OLMo.",
    "operationId": "crawler_ai2bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ai2bot-dolma.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Ai2Bot-Dolma",
    "description": "The variant of AI2's crawler named for the Dolma corpus specifically.",
    "operationId": "crawler_ai2bot_dolma",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/cohere-ai.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for cohere-ai",
    "description": "Cohere's fetcher, used when its assistant products need a page.",
    "operationId": "crawler_cohere_ai",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/cohere-training-data-crawler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for cohere-training-data-crawler",
    "description": "Cohere's separately-named bulk crawler for model training data, split out so consent for training and consent for retrieval can differ.",
    "operationId": "crawler_cohere_training_data_crawler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/mistralai-user.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for MistralAI-User",
    "description": "Fetches a page when a Le Chat user asks Mistral's assistant to read it.",
    "operationId": "crawler_mistralai_user",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/youbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YouBot",
    "description": "You.com's crawler, feeding its AI search product and its search API.",
    "operationId": "crawler_youbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/diffbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Diffbot",
    "description": "Extracts structured records from pages to build a commercial knowledge graph that is resold and used for retrieval and training.",
    "operationId": "crawler_diffbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/omgilibot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for omgilibot",
    "description": "Webz.io's crawler, collecting web and forum text sold as datasets, including to model builders.",
    "operationId": "crawler_omgilibot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/omgili.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for omgili",
    "description": "The older robots token for the same Webz.io collection, still honoured and still worth listing.",
    "operationId": "crawler_omgili",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/webzio-extended.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Webzio-Extended",
    "description": "Webz.io's opt-out token specifically for AI training reuse, in the pattern Google and Apple established.",
    "operationId": "crawler_webzio_extended",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/imagesiftbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ImagesiftBot",
    "description": "Crawls images for Hive AI's reverse-image and dataset products. Image-heavy sites see this one long before they see the text crawlers.",
    "operationId": "crawler_imagesiftbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/timpibot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Timpibot",
    "description": "A distributed crawler building an independent search index outside the Google/Bing duopoly.",
    "operationId": "crawler_timpibot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot",
    "description": "Semrush's backlink and keyword crawler. It is not an AI crawler, but it is usually in the top three by volume on any site, and it is the cheapest block on this list.",
    "operationId": "crawler_semrushbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-ocob.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-OCOB",
    "description": "Semrush's separately-tokenised crawler for its AI content tooling, split out so SEO crawling and AI reuse can be answered differently.",
    "operationId": "crawler_semrushbot_ocob",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ahrefsbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AhrefsBot",
    "description": "Ahrefs' backlink crawler, and one of the largest non-search crawlers on the web by request volume.",
    "operationId": "crawler_ahrefsbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/archive-org-bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for archive.org_bot",
    "description": "The Wayback Machine's crawler. Preservation rather than AI, but it lands in the same 'is this bot welcome' decision and its output is a public corpus.",
    "operationId": "crawler_archive_org_bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ia-archiver.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ia_archiver",
    "description": "The legacy Alexa/Internet Archive token, still present in most robots.txt files and still occasionally honoured.",
    "operationId": "crawler_ia_archiver",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexBot",
    "description": "Yandex's search crawler, which also feeds Alice and Yandex's generative answers.",
    "operationId": "crawler_yandexbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/baiduspider.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Baiduspider",
    "description": "Baidu's search crawler, and the ingest path for Baidu's Ernie-backed answers.",
    "operationId": "crawler_baiduspider",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/seznambot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SeznamBot",
    "description": "Seznam's crawler — the dominant search engine in the Czech Republic and one of the few national engines with its own index.",
    "operationId": "crawler_seznambot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yeti.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Yeti",
    "description": "Naver's crawler. Naver is South Korea's largest search portal and runs its own index and its own generative answers.",
    "operationId": "crawler_yeti",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/petalbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for PetalBot",
    "description": "Huawei's crawler for Petal Search, shipped as the default search on Huawei devices.",
    "operationId": "crawler_petalbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/firecrawlagent.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for FirecrawlAgent",
    "description": "A hosted scrape-to-markdown service that LLM applications call to read pages. The requester is whoever is building on it, not Firecrawl itself, so volume and intent vary wildly.",
    "operationId": "crawler_firecrawlagent",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/scrapy.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Scrapy",
    "description": "Not an operator: the default user-agent of the most common Python crawling framework. Anyone can be behind it. Modern Scrapy obeys robots.txt by default, which is why the default UA is still worth a rule.",
    "operationId": "crawler_scrapy",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/img2dataset.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for img2dataset",
    "description": "The tool used to turn image-URL lists such as LAION's into downloaded training sets. It is run by whoever is building a dataset, not by a single operator.",
    "operationId": "crawler_img2dataset",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googlebot-video.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Googlebot-Video",
    "description": "The video half of Googlebot. It crawls video files and the pages around them for Google Video search, and it is matched by a robots.txt group for Googlebot as well as by its own token.",
    "operationId": "crawler_googlebot_video",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googleother-image.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GoogleOther-Image",
    "description": "The image variant of GoogleOther: one-off fetches by Google product and research teams that are not Search. It also answers to a GoogleOther group in robots.txt.",
    "operationId": "crawler_googleother_image",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googleother-video.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GoogleOther-Video",
    "description": "The video variant of GoogleOther, used for internal Google fetches that do not belong to Search.",
    "operationId": "crawler_googleother_video",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/apis-google.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for APIs-Google",
    "description": "Delivers push notifications for Google APIs to a webhook you registered. It is a special-case crawler: it ignores the robots.txt * group, because the fetch is a delivery to an address you asked it to deliver to.",
    "operationId": "crawler_apis_google",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/adsbot-google.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AdsBot-Google",
    "description": "Checks the quality of desktop landing pages for Google Ads. Google documents that it ignores the robots.txt * group with the ad publisher's permission, and obeys a group named for its own token.",
    "operationId": "crawler_adsbot_google",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/adsbot-google-mobile.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AdsBot-Google-Mobile",
    "description": "The mobile-web landing page checker for Google Ads. Same rules as AdsBot-Google: the * group does not apply to it, its own token does.",
    "operationId": "crawler_adsbot_google_mobile",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/adsbot-google-mobile-apps.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AdsBot-Google-Mobile-Apps",
    "description": "Checks Android app landing pages for Google Ads. It obeys a group named for its own token and, per Google, follows the AdsBot-Google rules otherwise.",
    "operationId": "crawler_adsbot_google_mobile_apps",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/mediapartners-google.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Mediapartners-Google",
    "description": "The AdSense crawler. It reads a page so AdSense can choose relevant ads for it, and it is a special-case crawler that ignores the robots.txt * group.",
    "operationId": "crawler_mediapartners_google",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-safety.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Safety",
    "description": "Google's abuse-investigation fetcher: malware review, phishing reports and similar. Google documents that it ignores robots.txt entirely, and a robots.txt rule for it does nothing.",
    "operationId": "crawler_google_safety",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/feedfetcher-google.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for FeedFetcher-Google",
    "description": "Crawls RSS and Atom feeds for Google News and WebSub. It is a user-triggered fetcher, and Google documents that those generally ignore robots.txt because a person asked for the fetch. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_feedfetcher_google",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-read-aloud.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Read-Aloud",
    "description": "Fetches a page so Google can read it out loud with text-to-speech, at the moment a user asks. Formerly google-speakr. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_google_read_aloud",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-site-verification.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Site-Verification",
    "description": "Fetches the token file or meta tag that proves you own a site, when you click verify in Search Console. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_google_site_verification",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-cws.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-CWS",
    "description": "The Chrome Web Store fetcher. It requests the URLs a developer put in the metadata of a Chrome extension or theme. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_google_cws",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-pinpoint.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Pinpoint",
    "description": "Fetches individual URLs that a Pinpoint user — usually a journalist or researcher — added as a source to their own document collection. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_google_pinpoint",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googleproducer.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GoogleProducer",
    "description": "Google Publisher Center: fetches the feeds a publisher explicitly supplied for Google News landing pages. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_googleproducer",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/googlemessages.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for GoogleMessages",
    "description": "Generates the link preview when somebody sends one of your URLs in Google Messages. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_googlemessages",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-gemininotebook.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-GeminiNotebook",
    "description": "Fetches a URL a Gemini Notebook (formerly NotebookLM) user added as a source to their notebook. The former agent string Google-NotebookLM is documented as supported until August 2026. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
    "operationId": "crawler_google_gemininotebook",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/google-agent.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Google-Agent",
    "description": "Agents hosted on Google infrastructure navigating the web and taking actions on a user's request. Google names one prefix list for it — user-triggered-agents.json — and is separately experimenting with Web Bot Auth under the identity https://agent.bot.goog.",
    "operationId": "crawler_google_agent",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandeximages.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexImages",
    "description": "Indexes images for Yandex Images. Yandex's robot table marks it as taking the general robots.txt rules into account.",
    "operationId": "crawler_yandeximages",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexvideo.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexVideo",
    "description": "Indexes video for Yandex video search. Obeys the general robots.txt rules per Yandex's own table.",
    "operationId": "crawler_yandexvideo",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexmedia.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexMedia",
    "description": "Indexes multimedia data for Yandex. Takes the general robots.txt rules into account.",
    "operationId": "crawler_yandexmedia",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexblogs.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexBlogs",
    "description": "Yandex's blog-search robot; it indexes post comments as well as posts.",
    "operationId": "crawler_yandexblogs",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexmarket.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexMarket",
    "description": "The robot behind Yandex Market, Yandex's shopping comparison service. Version 1.0 is documented as obeying the general rules; version 2.0 is documented as not.",
    "operationId": "crawler_yandexmarket",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexwebmaster.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexWebmaster",
    "description": "The fetcher behind Yandex Webmaster, the console a site owner uses to inspect their own site.",
    "operationId": "crawler_yandexwebmaster",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexmobilebot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexMobileBot",
    "description": "Decides whether a page's layout is suitable for mobile devices. Yandex's table marks it as NOT taking the general robots.txt rules into account, so a * group does not stop it — a group named YandexMobileBot does.",
    "operationId": "crawler_yandexmobilebot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexfavicons.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexFavicons",
    "description": "Downloads your favicon so Yandex can show it beside your result. Documented as not taking the general robots.txt rules into account.",
    "operationId": "crawler_yandexfavicons",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexcalendar.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexCalendar",
    "description": "Downloads calendar files a user subscribed to. Yandex notes these files are often in directories that are disallowed for indexing, which is why the general rules are not applied.",
    "operationId": "crawler_yandexcalendar",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexdirect.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexDirect",
    "description": "Reads the content of Yandex Advertising Network partner pages to work out their topic so relevant ads can be matched. Documented as not taking the general robots.txt rules into account.",
    "operationId": "crawler_yandexdirect",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexmetrika.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexMetrika",
    "description": "Yandex Metrica's own fetcher. Two of its versions — the 2.0 yabs01 availability checker and the 4.0 CSS cache for Webvisor — are documented in Yandex's table as not using robots.txt at all.",
    "operationId": "crawler_yandexmetrika",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexrenderresourcesbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexRenderResourcesBot",
    "description": "Loads the CSS, JavaScript and images Yandex needs to render a page. Yandex documents the exact rule: it ignores robots.txt for a resource when the HTML page using it is allowed, and does not fetch the resource when that page is disallowed.",
    "operationId": "crawler_yandexrenderresourcesbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexscreenshotbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexScreenshotBot",
    "description": "Takes a screenshot of a page. Documented as not taking the general robots.txt rules into account.",
    "operationId": "crawler_yandexscreenshotbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexadditional.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexAdditional",
    "description": "The token that controls whether already-indexed pages may appear in Search with Yandex AI answers. Yandex's table says it makes no indexing requests of its own — it exists so a site can opt out of the generative answer without leaving the index.",
    "operationId": "crawler_yandexadditional",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexadditionalbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexAdditionalBot",
    "description": "The second token Yandex publishes for the same AI-answers opt-out. Both names appear in Yandex's own robot list, so a robots.txt that names only one of them is half a policy.",
    "operationId": "crawler_yandexadditionalbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yandexcombot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YandexComBot",
    "description": "Indexes content for Yandex search in languages other than Russian. Yandex documents that it can index content when there is no explicit robot-specific restriction — a * group is not one.",
    "operationId": "crawler_yandexcombot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/siteauditbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SiteAuditBot",
    "description": "Semrush's Site Audit crawler: it walks a site a customer owns and reports technical SEO problems. Semrush names it as the token to block for that product.",
    "operationId": "crawler_siteauditbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-ba.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-BA",
    "description": "The Backlink Audit crawler. It re-checks links pointing at a customer's site, which means it lands on the sites doing the linking.",
    "operationId": "crawler_semrushbot_ba",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-si.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-SI",
    "description": "Fetches pages for the On Page SEO Checker and similar advisory tools.",
    "operationId": "crawler_semrushbot_si",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-swa.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-SWA",
    "description": "Checks whether a URL is reachable, for the SEO Writing Assistant. One request per URL a writer references, not a crawl.",
    "operationId": "crawler_semrushbot_swa",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/splitsignalbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SplitSignalBot",
    "description": "Runs SEO A/B tests on a customer's own site with the SplitSignal tool.",
    "operationId": "crawler_splitsignalbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-ft.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-FT",
    "description": "Fetches full text for the Plagiarism Checker and similar text-comparison tools.",
    "operationId": "crawler_semrushbot_ft",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/semrushbot-esi.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SemrushBot-ESI",
    "description": "The crawler for Semrush Enterprise Site Intelligence, the enterprise tier's own site analysis.",
    "operationId": "crawler_semrushbot_esi",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/ahrefssiteaudit.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AhrefsSiteAudit",
    "description": "Ahrefs' site-audit crawler, separate from AhrefsBot. Ahrefs documents that it obeys robots.txt by default, and that a verified site owner can ask for it to be allowed to ignore robots.txt on their own site so the audit can see disallowed sections.",
    "operationId": "crawler_ahrefssiteaudit",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/mj12bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for MJ12bot",
    "description": "Majestic's link-graph crawler, run as a distributed community project. Majestic states plainly that it cannot restrict the bot to a fixed set of addresses, and offers a pre-arranged ident string in the request headers instead.",
    "operationId": "crawler_mj12bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/dotbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for DotBot",
    "description": "Moz's crawler for Link Explorer. Moz documents that it respects robots.txt and that dotbot is the token to name.",
    "operationId": "crawler_dotbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/rogerbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for rogerbot",
    "description": "Moz's Campaign crawler, which audits a site its own owner registered. Moz states there is no IP range for it — identification is by user-agent only.",
    "operationId": "crawler_rogerbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/dataforseobot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for DataForSeoBot",
    "description": "Builds the backlink and SERP datasets DataForSEO resells through its API, so one crawl reaches many downstream tools.",
    "operationId": "crawler_dataforseobot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/serpstatbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for serpstatbot",
    "description": "Serpstat's backlink crawler. It documents support for Crawl-delay up to 20 seconds, including a delay set on the * group.",
    "operationId": "crawler_serpstatbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/barkrowler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Barkrowler",
    "description": "Babbar's crawler, which builds the link graph behind their French-market SEO tooling.",
    "operationId": "crawler_barkrowler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/screaming-frog-seo-spider.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Screaming Frog SEO Spider",
    "description": "Not an operator: desktop crawling software that anybody can point at any site. The default user-agent identifies the tool, not who is running it, and the operator of the moment is whoever pressed start.",
    "operationId": "crawler_screaming_frog_seo_spider",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/seokicks.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SEOkicks",
    "description": "A German backlink index. Its documentation names SEOkicks as the user-agent to use in robots.txt.",
    "operationId": "crawler_seokicks",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/mojeekbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for MojeekBot",
    "description": "Mojeek's crawler. Mojeek runs one of the few genuinely independent web indexes — not a front end over Bing or Google — so it is one of the few blocks that removes you from an index nobody else can put you back into. Its documentation states it obeys the first record whose User-Agent contains MojeekBot, falling back to *.",
    "operationId": "crawler_mojeekbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/kagibot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Kagibot",
    "description": "The crawler for Kagi, a paid, ad-free search engine with its own index and its own assistant.",
    "operationId": "crawler_kagibot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/qwantbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Qwantbot",
    "description": "Qwant's crawler. Qwant documents that the string Qwantbot always appears in its user-agents whatever the crawler version, which is what makes a substring match safe here.",
    "operationId": "crawler_qwantbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/qwantbot-news.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Qwantbot-news",
    "description": "The news variant of Qwant's crawler, documented alongside the main one and carrying the same Qwantbot substring.",
    "operationId": "crawler_qwantbot_news",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/slackbot-linkexpanding.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Slackbot-LinkExpanding",
    "description": "Fetches a page to build the unfurl card when somebody pastes your link into Slack. One paste, one fetch.",
    "operationId": "crawler_slackbot_linkexpanding",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/slackbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Slackbot",
    "description": "The other half of Slack's pair: the agent that reads robots.txt and handles Slack's non-unfurl fetches. Slack documents both strings on one page.",
    "operationId": "crawler_slackbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/pinterestbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Pinterestbot",
    "description": "Pinterest's crawler. It indexes pages so people can find them on Pinterest and re-reads product pages to keep price and title on a Pin current. Pinterest states that content it crawls is not used to train their Canvas image generation model.",
    "operationId": "crawler_pinterestbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/bedrockbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for bedrockbot",
    "description": "The web crawler an AWS customer points at URLs they chose, to build a knowledge base for a Bedrock application. AWS documents that it respects robots.txt and that the user-agent carries a per-customer suffix, so you can allow or refuse one customer's crawl by naming bedrockbot-UUID.",
    "operationId": "crawler_bedrockbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/cloudflare-autorag.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Cloudflare-AutoRAG",
    "description": "The crawler behind Cloudflare's AI Search / AutoRAG, which indexes a website into a retrieval index for an application. Cloudflare's own documentation warns that a bot-blocking rule on your zone will also stop this crawler and tells you to allow-list it.",
    "operationId": "crawler_cloudflare_autorag",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/exasearchbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ExaSearchBot",
    "description": "Exa's crawler. It discovers and indexes public pages so they can be retrieved and cited through Exa's search API, which is one of the common retrieval backends behind agent frameworks.",
    "operationId": "crawler_exasearchbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/shapbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ShapBot",
    "description": "Parallel's crawler. It collects and structures web content to power the search, extraction and deep-research APIs that Parallel sells to agent builders.",
    "operationId": "crawler_shapbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/terracotta.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for TerraCotta",
    "description": "Ceramic AI's crawler, which indexes public content for a web-scale search API aimed at LLMs and agents.",
    "operationId": "crawler_terracotta",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/crawlspace.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Crawlspace",
    "description": "A crawling platform: customers run their own crawls on it to feed agents, RAG pipelines and structured-data workflows. Like Firecrawl, the party behind any given request is the customer, not the platform.",
    "operationId": "crawler_crawlspace",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/panscient.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Panscient",
    "description": "Compiles structured data about businesses and business professionals using machine learning. Panscient's FAQ states it obeys robots.txt.",
    "operationId": "crawler_panscient",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/sbintuitionsbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for SBIntuitionsBot",
    "description": "SB Intuitions is SoftBank's Japanese LLM lab; this crawler gathers data used in that model development and in information analysis. The operator publishes a dedicated bot page.",
    "operationId": "crawler_sbintuitionsbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/icc-crawler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ICC-Crawler",
    "description": "Operated by NICT, Japan's national information and communications research institute. The collected data supports AI research and, per the operator, is also provided to third parties including commercial companies.",
    "operationId": "crawler_icc_crawler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/cotoyogi.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Cotoyogi",
    "description": "A crawler run by ROIS-DS, a Japanese inter-university research organisation, collecting Japanese-language text for AI training. It publishes a crawler page in English and Japanese.",
    "operationId": "crawler_cotoyogi",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/isscyberriskcrawler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ISSCyberRiskCrawler",
    "description": "Crawls in order to train models that score a company's cyber risk. The ai.robots.txt dataset records the operator as not respecting robots.txt; ISS publishes no compliance statement of its own.",
    "operationId": "crawler_isscyberriskcrawler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/sidetrade-indexer-bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Sidetrade indexer bot",
    "description": "Sidetrade extracts web data for a range of uses including training its AI products for order-to-cash and customer-data work.",
    "operationId": "crawler_sidetrade_indexer_bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/yak.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for YaK",
    "description": "Meltwater's crawler, feeding the live data stream behind its media-monitoring and consumer-intelligence suite.",
    "operationId": "crawler_yak",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/atlassian-bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for atlassian-bot",
    "description": "Indexes a website so it can be searched and cited by Rovo, Atlassian's generative assistant inside Jira and Confluence. Atlassian's documentation walks a customer through editing robots.txt for it, which is as close to a compliance statement as this list gets.",
    "operationId": "crawler_atlassian_bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/klaviyoaibot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for KlaviyoAIBot",
    "description": "Fetches pages from domains a Klaviyo customer has explicitly connected to their own account, to power Klaviyo's Kai customer agent. It is scoped to connected domains rather than the open web.",
    "operationId": "crawler_klaviyoaibot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/quillbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for QuillBot",
    "description": "Operated by QuillBot as part of its writing, paraphrasing and AI-detection products. The dataset also records a second token, quillbot.com, for the same operator.",
    "operationId": "crawler_quillbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/phindbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for PhindBot",
    "description": "Phind is an answer engine for developers that combines live web search with its own models. This is the crawler behind those answers.",
    "operationId": "crawler_phindbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/andibot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Andibot",
    "description": "The crawler for Andi, a small generative search assistant that summarises pages rather than listing them.",
    "operationId": "crawler_andibot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/anomura.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Anomura",
    "description": "Direqt's search crawler. It indexes the sites of Direqt's own publisher customers so their on-site chatbots can answer from them.",
    "operationId": "crawler_anomura",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/aiwebindex.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AIWebIndex",
    "description": "Builds an index of public pages and serves them to AI agents as extracted readable text, with attribution and a link back. Lyrenth publishes a crawler policy stating it does not train foundation models on what it collects and that it obeys robots.txt.",
    "operationId": "crawler_aiwebindex",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/factset-spyderbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Factset_spyderbot",
    "description": "FactSet's crawler, collecting data used in AI model training for its financial data and analytics products.",
    "operationId": "crawler_factset_spyderbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/poseidon-research-crawler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Poseidon Research Crawler",
    "description": "A crawler run by Poseidon Research, a lab working on interpretability research for AI systems.",
    "operationId": "crawler_poseidon_research_crawler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/qualifiedbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for QualifiedBot",
    "description": "Analyses a customer's website so Qualified's AI sales chatbots can answer questions about it in context.",
    "operationId": "crawler_qualifiedbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/reflectionbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Reflectionbot",
    "description": "An undocumented crawler whose user-agent links to Reflection AI, a company building AI models. The link in the user-agent is the only public statement of purpose that exists.",
    "operationId": "crawler_reflectionbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/thinkbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Thinkbot",
    "description": "Collects pages for analysis of how sites are adopting AI and automation. The ai.robots.txt dataset records the operator as not respecting robots.txt.",
    "operationId": "crawler_thinkbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/aihitbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for aiHitBot",
    "description": "aiHit's automated collector, building a company dataset from public company websites.",
    "operationId": "crawler_aihitbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/linguee-bot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Linguee Bot",
    "description": "Gathers bilingual text for Linguee's translation corpus and the machine translation trained on it. Recorded in the ai.robots.txt dataset as not respecting robots.txt.",
    "operationId": "crawler_linguee_bot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/lightpanda.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Lightpanda",
    "description": "A purpose-built headless browser for AI and automation — a runtime, not an operator. Whether robots.txt is honoured is left to whoever runs it, which is what its maintainers say themselves.",
    "operationId": "crawler_lightpanda",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/laiondownloader.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for LAIONDownloader",
    "description": "LAION's downloader, used to materialise the image and text datasets the non-profit publishes for machine-learning research. LAION's own FAQ is the source for its robots.txt position.",
    "operationId": "crawler_laiondownloader",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/velenpublicwebcrawler.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for VelenPublicWebCrawler",
    "description": "Hunter's crawler, written in Go, building business datasets and machine-learning models from public pages. Its page states it follows robots.txt and meta directives and never fetches more than one page every two seconds.",
    "operationId": "crawler_velenpublicwebcrawler",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/awariosmartbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AwarioSmartBot",
    "description": "Awario's brand-monitoring crawler. It documents one request per three seconds, honours Crawl-delay, and states it does not use consecutive IP blocks so identification is by user-agent only.",
    "operationId": "crawler_awariosmartbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/awariorssbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for AwarioRssBot",
    "description": "The feed-reading half of Awario's pair, documented on the same page and under the same crawl-rate policy.",
    "operationId": "crawler_awariorssbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/echoboxbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for EchoboxBot",
    "description": "Collects data supporting Echobox's AI-driven social and email distribution products, which publishers use to schedule and target their own content.",
    "operationId": "crawler_echoboxbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/meta-webindexer.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Meta-WebIndexer",
    "description": "Per Meta's crawler documentation, Meta-WebIndexer navigates the web to improve the quality of Meta AI's search results. It is a third Meta token alongside Meta-ExternalAgent and Meta-ExternalFetcher, and the newest of them.",
    "operationId": "crawler_meta_webindexer",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/chatgpt-agent.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for ChatGPT Agent",
    "description": "ChatGPT's agent mode driving a real browser: it navigates and interacts with sites to finish a multi-step task a user gave it. OpenAI governs it with the ChatGPT-User token and the ChatGPT-User prefix list rather than a token of its own, so the robots rule and the address check are the same ones.",
    "operationId": "crawler_chatgpt_agent",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/wpbot.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for wpbot",
    "description": "Supports the AI Chatbot for WordPress plugin: it reads pages so the plugin can answer from a site's own content. The operator provides an opt-out through a form rather than through robots.txt.",
    "operationId": "crawler_wpbot",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  },
  "/crawler/crawl4ai.json": {
   "get": {
    "tags": [
     "crawlers"
    ],
    "summary": "Record for Crawl4AI",
    "description": "An open-source LLM-oriented crawler and scraper library, run by whoever installs it. Like Scrapy, the default user-agent identifies the software and says nothing about who is behind the request.",
    "operationId": "crawler_crawl4ai",
    "produces": [
     "application/json"
    ],
    "responses": {
     "200": {
      "description": "The crawler record",
      "headers": {
       "Link": {
        "type": "string",
        "description": "RFC 8288 links: the changes feed, a sibling document and the OpenAPI description of this host."
       }
      },
      "schema": {
       "$ref": "#/definitions/Crawler"
      }
     }
    }
   }
  }
 },
 "tags": [
  {
   "name": "crawlers",
   "description": "One record per crawler."
  },
  {
   "name": "bulk",
   "description": "The whole dataset in several shapes."
  },
  {
   "name": "ip-ranges",
   "description": "Operator-published prefixes, normalised."
  },
  {
   "name": "robots",
   "description": "Ready-made robots.txt policy files."
  },
  {
   "name": "status",
   "description": "Freshness of the upstream sources."
  },
  {
   "name": "changes",
   "description": "What changed since your last read, with a cursor to send back. The cheap way to stay current without re-downloading anything."
  },
  {
   "name": "tools",
   "description": "The read-only tools this host runs over MCP, as keyless GET endpoints. Same implementation, no JSON-RPC session required."
  }
 ],
 "definitions": {
  "Change": {
   "type": "object",
   "description": "One observed change. `kind` distinguishes what really happened: ip_ranges_changed means the prefix SET moved and the added/removed lists are the real difference; upstream_republished means the operator reissued the same prefixes, which is not a reason to re-download; source_failed means an upstream stopped answering and its last good prefixes are still being served, marked stale.",
   "required": [
    "seq",
    "at",
    "kind"
   ],
   "properties": {
    "seq": {
     "type": "integer",
     "description": "Monotonic cursor."
    },
    "at": {
     "type": "string",
     "format": "date-time",
     "description": "When WE observed it — the fetch that saw the change."
    },
    "kind": {
     "type": "string",
     "enum": [
      "ip_ranges_changed",
      "upstream_republished",
      "source_failed",
      "source_recovered",
      "source_added",
      "source_removed",
      "crawler_added",
      "crawler_changed",
      "crawler_removed",
      "baseline"
     ]
    },
    "title": {
     "type": "string"
    },
    "detail": {
     "type": "string"
    },
    "source": {
     "type": "string",
     "description": "Slug of the IP-range source."
    },
    "crawler": {
     "type": "string",
     "description": "Slug of the crawler record."
    },
    "operator_published_at": {
     "type": "string",
     "description": "The operator's own creationTime, carried through untouched."
    },
    "added_count": {
     "type": "integer"
    },
    "removed_count": {
     "type": "integer"
    },
    "ipv4_added": {
     "type": "array",
     "items": {
      "type": "string"
     }
    },
    "ipv4_removed": {
     "type": "array",
     "items": {
      "type": "string"
     }
    },
    "ipv6_added": {
     "type": "array",
     "items": {
      "type": "string"
     }
    },
    "ipv6_removed": {
     "type": "array",
     "items": {
      "type": "string"
     }
    },
    "lists_truncated": {
     "type": "boolean",
     "description": "Counts are always exact; when true the prefix LISTS are a sample and the full set is in the source document."
    },
    "affects": {
     "type": "array",
     "items": {
      "type": "string"
     },
     "description": "The documents this change moved."
    }
   }
  },
  "Crawler": {
   "type": "object",
   "required": [
    "slug",
    "name",
    "operator",
    "category",
    "robots_token"
   ],
   "properties": {
    "slug": {
     "type": "string",
     "description": "Stable identifier used in URLs."
    },
    "name": {
     "type": "string"
    },
    "operator": {
     "type": "string"
    },
    "operator_slug": {
     "type": "string"
    },
    "category": {
     "type": "string",
     "enum": [
      "ai-search",
      "ai-training",
      "archive",
      "dataset",
      "preview",
      "search",
      "seo",
      "tool",
      "user-fetch"
     ]
    },
    "robots_token": {
     "type": "string",
     "description": "Exact User-agent value for robots.txt."
    },
    "user_agent_substring": {
     "type": "string",
     "description": "Substring that reliably identifies it in a UA header. A match is a claim, not a proof."
    },
    "user_agent_example": {
     "type": "string"
    },
    "respects_robots_txt": {
     "type": "string",
     "enum": [
      "documented",
      "by-design-no",
      "disputed",
      "n-a"
     ]
    },
    "verification_method": {
     "type": "string",
     "enum": [
      "published-ranges",
      "reverse-dns",
      "none"
     ]
    },
    "published_ip_ranges_url": {
     "type": "string",
     "x-nullable": true,
     "format": "uri"
    },
    "ipv4_prefix_count": {
     "type": "integer"
    },
    "ipv6_prefix_count": {
     "type": "integer"
    },
    "what_it_is": {
     "type": "string"
    },
    "cost_of_blocking": {
     "type": "string",
     "description": "What you lose by disallowing it. This index's own assessment, not the operator's."
    },
    "operator_docs": {
     "type": "string",
     "format": "uri"
    },
    "last_reviewed": {
     "type": "string",
     "format": "date"
    }
   }
  }
 }
}