curl -s https://www.pathwren.workers.dev/snippet/nginx-map.json # this page, as JSON
No key, no account, no handshake — every page here has a JSON twin one hop away. Machine doors: 6 keyless GET tools · documents.json · changes · llms.txt · openapi.json · agent card · mcp · a2a
A map block that turns the user-agent into a category, so you can rate-limit training crawlers, allow AI search, and 403 the disputed ones without maintaining three separate if-chains.
curl -s https://www.pathwren.workers.dev/snippet/nginx-map.txt -o nginx-map.conf
# AI Crawler Index — nginx classification map
# curl -s https://www.pathwren.workers.dev/snippet/nginx-map.txt
map $http_user_agent $ai_crawler {
default "";
"~*AdsBot\-Google" "tool";
"~*AdsBot\-Google\-Mobile" "tool";
"~*AdsBot\-Google\-Mobile\-Apps" "tool";
"~*AhrefsBot" "seo";
"~*AhrefsSiteAudit" "seo";
"~*AI2Bot" "dataset";
"~*Ai2Bot\-Dolma" "dataset";
"~*aiHitBot" "dataset";
"~*AIWebIndex" "ai-search";
"~*Amazonbot" "ai-search";
"~*Andibot" "ai-search";
"~*Anomura" "ai-search";
"~*anthropic\-ai" "ai-training";
"~*APIs\-Google" "tool";
"~*Applebot" "search";
"~*archive\.org_bot" "archive";
"~*atlassian\-bot" "ai-search";
"~*AwarioRssBot" "dataset";
"~*AwarioSmartBot" "dataset";
"~*Baiduspider" "search";
"~*barkrowler" "seo";
"~*bedrockbot" "ai-search";
"~*bingbot" "search";
"~*Bytespider" "ai-training";
"~*CCBot" "dataset";
"~*ChatGPT\ Agent" "user-fetch";
"~*ChatGPT\-User" "user-fetch";
"~*Claude\-SearchBot" "ai-search";
"~*Claude\-User" "user-fetch";
"~*Claude\-Web" "ai-search";
"~*ClaudeBot" "ai-training";
"~*Cloudflare\-AutoRAG" "ai-search";
"~*cohere\-ai" "user-fetch";
"~*cohere\-training\-data\-crawler" "ai-training";
"~*Cotoyogi" "ai-training";
"~*Crawl4AI" "tool";
"~*Crawlspace" "tool";
"~*DataForSeoBot" "seo";
"~*Diffbot" "dataset";
"~*dotbot" "seo";
"~*DuckAssistBot" "ai-search";
"~*DuckDuckBot" "search";
"~*EchoboxBot" "dataset";
"~*ExaSearchBot" "ai-search";
"~*FacebookBot" "ai-training";
"~*facebookexternalhit" "preview";
"~*Factset_spyderbot" "ai-training";
"~*FeedFetcher\-Google" "tool";
"~*FirecrawlAgent" "tool";
"~*Google\-Agent" "user-fetch";
"~*Google\-CloudVertexBot" "ai-search";
"~*Google\-CWS" "tool";
"~*Google\-GeminiNotebook" "user-fetch";
"~*Google\-InspectionTool" "tool";
"~*Google\-Pinpoint" "user-fetch";
"~*Google\-Read\-Aloud" "user-fetch";
"~*Google\-Safety" "tool";
"~*Google\-Site\-Verification" "tool";
"~*Googlebot" "search";
"~*Googlebot\-Image" "search";
"~*Googlebot\-News" "search";
"~*Googlebot\-Video" "search";
"~*GoogleMessages" "preview";
"~*GoogleOther" "ai-training";
"~*GoogleOther\-Image" "ai-training";
"~*GoogleOther\-Video" "ai-training";
"~*GoogleProducer" "tool";
"~*GPTBot" "ai-training";
"~*ia_archiver" "archive";
"~*ICC\-Crawler" "ai-training";
"~*ImagesiftBot" "dataset";
"~*img2dataset" "dataset";
"~*ISSCyberRiskCrawler" "ai-training";
"~*Kagibot" "search";
"~*KlaviyoAIBot" "ai-search";
"~*LAIONDownloader" "dataset";
"~*Lightpanda" "tool";
"~*Linguee\ Bot" "ai-training";
"~*Mediapartners\-Google" "tool";
"~*meta\-externalagent" "ai-training";
"~*meta\-externalfetcher" "user-fetch";
"~*Meta\-WebIndexer" "ai-search";
"~*MistralAI\-User" "user-fetch";
"~*MJ12bot" "seo";
"~*MojeekBot" "search";
"~*OAI\-SearchBot" "ai-search";
"~*omgili" "dataset";
"~*omgilibot" "dataset";
"~*panscient\.com" "dataset";
"~*Perplexity\-User" "user-fetch";
"~*PerplexityBot" "ai-search";
"~*PetalBot" "search";
"~*PhindBot" "ai-search";
"~*Pinterestbot" "search";
"~*Poseidon\ Research\ Crawler" "ai-training";
"~*QualifiedBot" "ai-search";
"~*QuillBot" "ai-training";
"~*Qwantbot" "search";
"~*Qwantbot\-news" "search";
"~*Reflectionbot" "ai-training";
"~*rogerbot" "seo";
"~*SBIntuitionsBot" "ai-training";
"~*Scrapy" "tool";
"~*Screaming\ Frog\ SEO\ Spider" "tool";
"~*SemrushBot" "seo";
"~*SemrushBot\-BA" "seo";
"~*SemrushBot\-ESI" "seo";
"~*SemrushBot\-FT" "seo";
"~*SemrushBot\-OCOB" "ai-training";
"~*SemrushBot\-SI" "seo";
"~*SemrushBot\-SWA" "seo";
"~*SEOkicks" "seo";
"~*serpstatbot" "seo";
"~*SeznamBot" "search";
"~*ShapBot" "ai-search";
"~*Sidetrade\ indexer\ bot" "ai-training";
"~*SiteAuditBot" "seo";
"~*Slackbot" "preview";
"~*Slackbot\-LinkExpanding" "preview";
"~*SplitSignalBot" "seo";
"~*Storebot\-Google" "search";
"~*TerraCotta" "ai-search";
"~*Thinkbot" "dataset";
"~*TikTokSpider" "ai-training";
"~*Timpibot" "search";
"~*VelenPublicWebCrawler" "dataset";
"~*Webzio\-Extended" "ai-training";
"~*wpbot" "tool";
"~*YaK" "dataset";
"~*YandexAdditional" "ai-training";
"~*YandexAdditionalBot" "ai-training";
"~*YandexBlogs" "search";
"~*YandexBot" "search";
"~*YandexCalendar" "user-fetch";
"~*YandexComBot" "search";
"~*YandexDirect" "tool";
"~*YandexFavicons" "search";
"~*YandexImages" "search";
"~*YandexMarket" "search";
"~*YandexMedia" "search";
"~*YandexMetrika" "tool";
"~*YandexMobileBot" "search";
"~*YandexRenderResourcesBot" "search";
"~*YandexScreenshotBot" "tool";
"~*YandexVideo" "search";
"~*YandexWebmaster" "tool";
"~*Yeti" "search";
"~*YouBot" "ai-search";
}
# Then, in a server block: log it, rate-limit it, or refuse it.
# access_log /var/log/nginx/ai.log combined if=$ai_crawler;
# if ($ai_crawler = "disputed") { return 403; }