{
 "tool": "ai-access",
 "endpoint": "https://www.pathwren.workers.dev/tools/ai-access",
 "asked": {
  "robots_txt": "User-agent: GPTBot\nDisallow: /\n",
  "path": "/"
 },
 "summary": "Of 150 AI crawlers in this index, this robots.txt blocks 1 at / and allows 149.\n\nBLOCKED: GPTBot\n\nALLOWED: OAI-SearchBot, ChatGPT-User, ClaudeBot, Claude-SearchBot, Claude-User, anthropic-ai, Claude-Web, Google-Extended, Googlebot, GoogleOther, Google-CloudVertexBot, Google-InspectionTool, Googlebot-Image, Googlebot-News, Storebot-Google, bingbot, Applebot, Applebot-Extended, PerplexityBot, Perplexity-User, CCBot, Bytespider, TikTokSpider, meta-externalagent, meta-externalfetcher, facebookexternalhit, FacebookBot, Amazonbot, DuckAssistBot, DuckDuckBot, AI2Bot, Ai2Bot-Dolma, cohere-ai, cohere-training-data-crawler, MistralAI-User, YouBot, Diffbot, omgilibot, omgili, Webzio-Extended, ImagesiftBot, Timpibot, SemrushBot, SemrushBot-OCOB, AhrefsBot, archive.org_bot, ia_archiver, YandexBot, Baiduspider, SeznamBot, Yeti, PetalBot, FirecrawlAgent, Scrapy, img2dataset, Googlebot-Video, GoogleOther-Image, GoogleOther-Video, APIs-Google, AdsBot-Google, AdsBot-Google-Mobile, AdsBot-Google-Mobile-Apps, Mediapartners-Google, Google-Safety, FeedFetcher-Google, Google-Read-Aloud, Google-Site-Verification, Google-CWS, Google-Pinpoint, GoogleProducer, GoogleMessages, Google-GeminiNotebook, Google-Agent, YandexImages, YandexVideo, YandexMedia, YandexBlogs, YandexMarket, YandexWebmaster, YandexMobileBot, YandexFavicons, YandexCalendar, YandexDirect, YandexMetrika, YandexRenderResourcesBot, YandexScreenshotBot, YandexAdditional, YandexAdditionalBot, YandexComBot, SiteAuditBot, SemrushBot-BA, SemrushBot-SI, SemrushBot-SWA, SplitSignalBot, SemrushBot-FT, SemrushBot-ESI, AhrefsSiteAudit, MJ12bot, DotBot, rogerbot, DataForSeoBot, serpstatbot, Barkrowler, Screaming Frog SEO Spider, SEOkicks, MojeekBot, Kagibot, Qwantbot, Qwantbot-news, Slackbot-LinkExpanding, Slackbot, Pinterestbot, bedrockbot, Cloudflare-AutoRAG, ExaSearchBot, ShapBot, TerraCotta, Crawlspace, Panscient, SBIntuitionsBot, ICC-Crawler, Cotoyogi, ISSCyberRiskCrawler, Sidetrade indexer bot, YaK, atlassian-bot, KlaviyoAIBot, QuillBot, PhindBot, Andibot, Anomura, AIWebIndex, Factset_spyderbot, Poseidon Research Crawler, QualifiedBot, Reflectionbot, Thinkbot, aiHitBot, Linguee Bot, Lightpanda, LAIONDownloader, VelenPublicWebCrawler, AwarioSmartBot, AwarioRssBot, EchoboxBot, Meta-WebIndexer, ChatGPT Agent, wpbot, Crawl4AI\n",
 "answer": {
  "path_tested": "/",
  "crawlers_checked": 150,
  "blocked": 1,
  "allowed": 149,
  "by_category": {
   "ai-training": {
    "blocked": 1,
    "allowed": 26
   },
   "ai-search": {
    "blocked": 0,
    "allowed": 21
   },
   "user-fetch": {
    "blocked": 0,
    "allowed": 12
   },
   "search": {
    "blocked": 0,
    "allowed": 28
   },
   "tool": {
    "blocked": 0,
    "allowed": 22
   },
   "dataset": {
    "blocked": 0,
    "allowed": 17
   },
   "preview": {
    "blocked": 0,
    "allowed": 4
   },
   "seo": {
    "blocked": 0,
    "allowed": 17
   },
   "archive": {
    "blocked": 0,
    "allowed": 2
   }
  },
  "blocked_crawlers": [
   {
    "slug": "gptbot",
    "name": "GPTBot",
    "operator": "OpenAI",
    "decided_by": "Disallow: /",
    "named_explicitly": true
   }
  ],
  "allowed_crawlers": [
   {
    "slug": "oai-searchbot",
    "name": "OAI-SearchBot",
    "operator": "OpenAI",
    "category": "ai-search"
   },
   {
    "slug": "chatgpt-user",
    "name": "ChatGPT-User",
    "operator": "OpenAI",
    "category": "user-fetch"
   },
   {
    "slug": "claudebot",
    "name": "ClaudeBot",
    "operator": "Anthropic",
    "category": "ai-training"
   },
   {
    "slug": "claude-searchbot",
    "name": "Claude-SearchBot",
    "operator": "Anthropic",
    "category": "ai-search"
   },
   {
    "slug": "claude-user",
    "name": "Claude-User",
    "operator": "Anthropic",
    "category": "user-fetch"
   },
   {
    "slug": "anthropic-ai",
    "name": "anthropic-ai",
    "operator": "Anthropic",
    "category": "ai-training"
   },
   {
    "slug": "claude-web",
    "name": "Claude-Web",
    "operator": "Anthropic",
    "category": "ai-search"
   },
   {
    "slug": "google-extended",
    "name": "Google-Extended",
    "operator": "Google",
    "category": "ai-training"
   },
   {
    "slug": "googlebot",
    "name": "Googlebot",
    "operator": "Google",
    "category": "search"
   },
   {
    "slug": "googleother",
    "name": "GoogleOther",
    "operator": "Google",
    "category": "ai-training"
   },
   {
    "slug": "google-cloudvertexbot",
    "name": "Google-CloudVertexBot",
    "operator": "Google",
    "category": "ai-search"
   },
   {
    "slug": "google-inspectiontool",
    "name": "Google-InspectionTool",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "googlebot-image",
    "name": "Googlebot-Image",
    "operator": "Google",
    "category": "search"
   },
   {
    "slug": "googlebot-news",
    "name": "Googlebot-News",
    "operator": "Google",
    "category": "search"
   },
   {
    "slug": "storebot-google",
    "name": "Storebot-Google",
    "operator": "Google",
    "category": "search"
   },
   {
    "slug": "bingbot",
    "name": "bingbot",
    "operator": "Microsoft",
    "category": "search"
   },
   {
    "slug": "applebot",
    "name": "Applebot",
    "operator": "Apple",
    "category": "search"
   },
   {
    "slug": "applebot-extended",
    "name": "Applebot-Extended",
    "operator": "Apple",
    "category": "ai-training"
   },
   {
    "slug": "perplexitybot",
    "name": "PerplexityBot",
    "operator": "Perplexity",
    "category": "ai-search"
   },
   {
    "slug": "perplexity-user",
    "name": "Perplexity-User",
    "operator": "Perplexity",
    "category": "user-fetch"
   },
   {
    "slug": "ccbot",
    "name": "CCBot",
    "operator": "Common Crawl",
    "category": "dataset"
   },
   {
    "slug": "bytespider",
    "name": "Bytespider",
    "operator": "ByteDance",
    "category": "ai-training"
   },
   {
    "slug": "tiktokspider",
    "name": "TikTokSpider",
    "operator": "ByteDance",
    "category": "ai-training"
   },
   {
    "slug": "meta-externalagent",
    "name": "meta-externalagent",
    "operator": "Meta",
    "category": "ai-training"
   },
   {
    "slug": "meta-externalfetcher",
    "name": "meta-externalfetcher",
    "operator": "Meta",
    "category": "user-fetch"
   },
   {
    "slug": "facebookexternalhit",
    "name": "facebookexternalhit",
    "operator": "Meta",
    "category": "preview"
   },
   {
    "slug": "facebookbot",
    "name": "FacebookBot",
    "operator": "Meta",
    "category": "ai-training"
   },
   {
    "slug": "amazonbot",
    "name": "Amazonbot",
    "operator": "Amazon",
    "category": "ai-search"
   },
   {
    "slug": "duckassistbot",
    "name": "DuckAssistBot",
    "operator": "DuckDuckGo",
    "category": "ai-search"
   },
   {
    "slug": "duckduckbot",
    "name": "DuckDuckBot",
    "operator": "DuckDuckGo",
    "category": "search"
   },
   {
    "slug": "ai2bot",
    "name": "AI2Bot",
    "operator": "Allen Institute for AI",
    "category": "dataset"
   },
   {
    "slug": "ai2bot-dolma",
    "name": "Ai2Bot-Dolma",
    "operator": "Allen Institute for AI",
    "category": "dataset"
   },
   {
    "slug": "cohere-ai",
    "name": "cohere-ai",
    "operator": "Cohere",
    "category": "user-fetch"
   },
   {
    "slug": "cohere-training-data-crawler",
    "name": "cohere-training-data-crawler",
    "operator": "Cohere",
    "category": "ai-training"
   },
   {
    "slug": "mistralai-user",
    "name": "MistralAI-User",
    "operator": "Mistral AI",
    "category": "user-fetch"
   },
   {
    "slug": "youbot",
    "name": "YouBot",
    "operator": "You.com",
    "category": "ai-search"
   },
   {
    "slug": "diffbot",
    "name": "Diffbot",
    "operator": "Diffbot",
    "category": "dataset"
   },
   {
    "slug": "omgilibot",
    "name": "omgilibot",
    "operator": "Webz.io",
    "category": "dataset"
   },
   {
    "slug": "omgili",
    "name": "omgili",
    "operator": "Webz.io",
    "category": "dataset"
   },
   {
    "slug": "webzio-extended",
    "name": "Webzio-Extended",
    "operator": "Webz.io",
    "category": "ai-training"
   },
   {
    "slug": "imagesiftbot",
    "name": "ImagesiftBot",
    "operator": "Hive AI",
    "category": "dataset"
   },
   {
    "slug": "timpibot",
    "name": "Timpibot",
    "operator": "Timpi",
    "category": "search"
   },
   {
    "slug": "semrushbot",
    "name": "SemrushBot",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-ocob",
    "name": "SemrushBot-OCOB",
    "operator": "Semrush",
    "category": "ai-training"
   },
   {
    "slug": "ahrefsbot",
    "name": "AhrefsBot",
    "operator": "Ahrefs",
    "category": "seo"
   },
   {
    "slug": "archive-org-bot",
    "name": "archive.org_bot",
    "operator": "Internet Archive",
    "category": "archive"
   },
   {
    "slug": "ia-archiver",
    "name": "ia_archiver",
    "operator": "Internet Archive",
    "category": "archive"
   },
   {
    "slug": "yandexbot",
    "name": "YandexBot",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "baiduspider",
    "name": "Baiduspider",
    "operator": "Baidu",
    "category": "search"
   },
   {
    "slug": "seznambot",
    "name": "SeznamBot",
    "operator": "Seznam",
    "category": "search"
   },
   {
    "slug": "yeti",
    "name": "Yeti",
    "operator": "Naver",
    "category": "search"
   },
   {
    "slug": "petalbot",
    "name": "PetalBot",
    "operator": "Huawei",
    "category": "search"
   },
   {
    "slug": "firecrawlagent",
    "name": "FirecrawlAgent",
    "operator": "Firecrawl",
    "category": "tool"
   },
   {
    "slug": "scrapy",
    "name": "Scrapy",
    "operator": "Scrapy project",
    "category": "tool"
   },
   {
    "slug": "img2dataset",
    "name": "img2dataset",
    "operator": "LAION / img2dataset",
    "category": "dataset"
   },
   {
    "slug": "googlebot-video",
    "name": "Googlebot-Video",
    "operator": "Google",
    "category": "search"
   },
   {
    "slug": "googleother-image",
    "name": "GoogleOther-Image",
    "operator": "Google",
    "category": "ai-training"
   },
   {
    "slug": "googleother-video",
    "name": "GoogleOther-Video",
    "operator": "Google",
    "category": "ai-training"
   },
   {
    "slug": "apis-google",
    "name": "APIs-Google",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "adsbot-google",
    "name": "AdsBot-Google",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "adsbot-google-mobile",
    "name": "AdsBot-Google-Mobile",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "adsbot-google-mobile-apps",
    "name": "AdsBot-Google-Mobile-Apps",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "mediapartners-google",
    "name": "Mediapartners-Google",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "google-safety",
    "name": "Google-Safety",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "feedfetcher-google",
    "name": "FeedFetcher-Google",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "google-read-aloud",
    "name": "Google-Read-Aloud",
    "operator": "Google",
    "category": "user-fetch"
   },
   {
    "slug": "google-site-verification",
    "name": "Google-Site-Verification",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "google-cws",
    "name": "Google-CWS",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "google-pinpoint",
    "name": "Google-Pinpoint",
    "operator": "Google",
    "category": "user-fetch"
   },
   {
    "slug": "googleproducer",
    "name": "GoogleProducer",
    "operator": "Google",
    "category": "tool"
   },
   {
    "slug": "googlemessages",
    "name": "GoogleMessages",
    "operator": "Google",
    "category": "preview"
   },
   {
    "slug": "google-gemininotebook",
    "name": "Google-GeminiNotebook",
    "operator": "Google",
    "category": "user-fetch"
   },
   {
    "slug": "google-agent",
    "name": "Google-Agent",
    "operator": "Google",
    "category": "user-fetch"
   },
   {
    "slug": "yandeximages",
    "name": "YandexImages",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexvideo",
    "name": "YandexVideo",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexmedia",
    "name": "YandexMedia",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexblogs",
    "name": "YandexBlogs",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexmarket",
    "name": "YandexMarket",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexwebmaster",
    "name": "YandexWebmaster",
    "operator": "Yandex",
    "category": "tool"
   },
   {
    "slug": "yandexmobilebot",
    "name": "YandexMobileBot",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexfavicons",
    "name": "YandexFavicons",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexcalendar",
    "name": "YandexCalendar",
    "operator": "Yandex",
    "category": "user-fetch"
   },
   {
    "slug": "yandexdirect",
    "name": "YandexDirect",
    "operator": "Yandex",
    "category": "tool"
   },
   {
    "slug": "yandexmetrika",
    "name": "YandexMetrika",
    "operator": "Yandex",
    "category": "tool"
   },
   {
    "slug": "yandexrenderresourcesbot",
    "name": "YandexRenderResourcesBot",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "yandexscreenshotbot",
    "name": "YandexScreenshotBot",
    "operator": "Yandex",
    "category": "tool"
   },
   {
    "slug": "yandexadditional",
    "name": "YandexAdditional",
    "operator": "Yandex",
    "category": "ai-training"
   },
   {
    "slug": "yandexadditionalbot",
    "name": "YandexAdditionalBot",
    "operator": "Yandex",
    "category": "ai-training"
   },
   {
    "slug": "yandexcombot",
    "name": "YandexComBot",
    "operator": "Yandex",
    "category": "search"
   },
   {
    "slug": "siteauditbot",
    "name": "SiteAuditBot",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-ba",
    "name": "SemrushBot-BA",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-si",
    "name": "SemrushBot-SI",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-swa",
    "name": "SemrushBot-SWA",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "splitsignalbot",
    "name": "SplitSignalBot",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-ft",
    "name": "SemrushBot-FT",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "semrushbot-esi",
    "name": "SemrushBot-ESI",
    "operator": "Semrush",
    "category": "seo"
   },
   {
    "slug": "ahrefssiteaudit",
    "name": "AhrefsSiteAudit",
    "operator": "Ahrefs",
    "category": "seo"
   },
   {
    "slug": "mj12bot",
    "name": "MJ12bot",
    "operator": "Majestic",
    "category": "seo"
   },
   {
    "slug": "dotbot",
    "name": "DotBot",
    "operator": "Moz",
    "category": "seo"
   },
   {
    "slug": "rogerbot",
    "name": "rogerbot",
    "operator": "Moz",
    "category": "seo"
   },
   {
    "slug": "dataforseobot",
    "name": "DataForSeoBot",
    "operator": "DataForSEO",
    "category": "seo"
   },
   {
    "slug": "serpstatbot",
    "name": "serpstatbot",
    "operator": "Serpstat",
    "category": "seo"
   },
   {
    "slug": "barkrowler",
    "name": "Barkrowler",
    "operator": "Babbar",
    "category": "seo"
   },
   {
    "slug": "screaming-frog-seo-spider",
    "name": "Screaming Frog SEO Spider",
    "operator": "Screaming Frog",
    "category": "tool"
   },
   {
    "slug": "seokicks",
    "name": "SEOkicks",
    "operator": "SEOkicks",
    "category": "seo"
   },
   {
    "slug": "mojeekbot",
    "name": "MojeekBot",
    "operator": "Mojeek",
    "category": "search"
   },
   {
    "slug": "kagibot",
    "name": "Kagibot",
    "operator": "Kagi",
    "category": "search"
   },
   {
    "slug": "qwantbot",
    "name": "Qwantbot",
    "operator": "Qwant",
    "category": "search"
   },
   {
    "slug": "qwantbot-news",
    "name": "Qwantbot-news",
    "operator": "Qwant",
    "category": "search"
   },
   {
    "slug": "slackbot-linkexpanding",
    "name": "Slackbot-LinkExpanding",
    "operator": "Slack",
    "category": "preview"
   },
   {
    "slug": "slackbot",
    "name": "Slackbot",
    "operator": "Slack",
    "category": "preview"
   },
   {
    "slug": "pinterestbot",
    "name": "Pinterestbot",
    "operator": "Pinterest",
    "category": "search"
   },
   {
    "slug": "bedrockbot",
    "name": "bedrockbot",
    "operator": "Amazon",
    "category": "ai-search"
   },
   {
    "slug": "cloudflare-autorag",
    "name": "Cloudflare-AutoRAG",
    "operator": "Cloudflare",
    "category": "ai-search"
   },
   {
    "slug": "exasearchbot",
    "name": "ExaSearchBot",
    "operator": "Exa",
    "category": "ai-search"
   },
   {
    "slug": "shapbot",
    "name": "ShapBot",
    "operator": "Parallel",
    "category": "ai-search"
   },
   {
    "slug": "terracotta",
    "name": "TerraCotta",
    "operator": "Ceramic AI",
    "category": "ai-search"
   },
   {
    "slug": "crawlspace",
    "name": "Crawlspace",
    "operator": "Crawlspace",
    "category": "tool"
   },
   {
    "slug": "panscient",
    "name": "Panscient",
    "operator": "Panscient",
    "category": "dataset"
   },
   {
    "slug": "sbintuitionsbot",
    "name": "SBIntuitionsBot",
    "operator": "SB Intuitions",
    "category": "ai-training"
   },
   {
    "slug": "icc-crawler",
    "name": "ICC-Crawler",
    "operator": "NICT",
    "category": "ai-training"
   },
   {
    "slug": "cotoyogi",
    "name": "Cotoyogi",
    "operator": "ROIS-DS",
    "category": "ai-training"
   },
   {
    "slug": "isscyberriskcrawler",
    "name": "ISSCyberRiskCrawler",
    "operator": "ISS Corporate Solutions",
    "category": "ai-training"
   },
   {
    "slug": "sidetrade-indexer-bot",
    "name": "Sidetrade indexer bot",
    "operator": "Sidetrade",
    "category": "ai-training"
   },
   {
    "slug": "yak",
    "name": "YaK",
    "operator": "Meltwater",
    "category": "dataset"
   },
   {
    "slug": "atlassian-bot",
    "name": "atlassian-bot",
    "operator": "Atlassian",
    "category": "ai-search"
   },
   {
    "slug": "klaviyoaibot",
    "name": "KlaviyoAIBot",
    "operator": "Klaviyo",
    "category": "ai-search"
   },
   {
    "slug": "quillbot",
    "name": "QuillBot",
    "operator": "QuillBot",
    "category": "ai-training"
   },
   {
    "slug": "phindbot",
    "name": "PhindBot",
    "operator": "Phind",
    "category": "ai-search"
   },
   {
    "slug": "andibot",
    "name": "Andibot",
    "operator": "Andi",
    "category": "ai-search"
   },
   {
    "slug": "anomura",
    "name": "Anomura",
    "operator": "Direqt",
    "category": "ai-search"
   },
   {
    "slug": "aiwebindex",
    "name": "AIWebIndex",
    "operator": "Lyrenth",
    "category": "ai-search"
   },
   {
    "slug": "factset-spyderbot",
    "name": "Factset_spyderbot",
    "operator": "FactSet",
    "category": "ai-training"
   },
   {
    "slug": "poseidon-research-crawler",
    "name": "Poseidon Research Crawler",
    "operator": "Poseidon Research",
    "category": "ai-training"
   },
   {
    "slug": "qualifiedbot",
    "name": "QualifiedBot",
    "operator": "Qualified",
    "category": "ai-search"
   },
   {
    "slug": "reflectionbot",
    "name": "Reflectionbot",
    "operator": "Reflection AI",
    "category": "ai-training"
   },
   {
    "slug": "thinkbot",
    "name": "Thinkbot",
    "operator": "Thinkbot",
    "category": "dataset"
   },
   {
    "slug": "aihitbot",
    "name": "aiHitBot",
    "operator": "aiHit",
    "category": "dataset"
   },
   {
    "slug": "linguee-bot",
    "name": "Linguee Bot",
    "operator": "Linguee",
    "category": "ai-training"
   },
   {
    "slug": "lightpanda",
    "name": "Lightpanda",
    "operator": "Lightpanda",
    "category": "tool"
   },
   {
    "slug": "laiondownloader",
    "name": "LAIONDownloader",
    "operator": "LAION / img2dataset",
    "category": "dataset"
   },
   {
    "slug": "velenpublicwebcrawler",
    "name": "VelenPublicWebCrawler",
    "operator": "Hunter (Velen)",
    "category": "dataset"
   },
   {
    "slug": "awariosmartbot",
    "name": "AwarioSmartBot",
    "operator": "Awario",
    "category": "dataset"
   },
   {
    "slug": "awariorssbot",
    "name": "AwarioRssBot",
    "operator": "Awario",
    "category": "dataset"
   },
   {
    "slug": "echoboxbot",
    "name": "EchoboxBot",
    "operator": "Echobox",
    "category": "dataset"
   },
   {
    "slug": "meta-webindexer",
    "name": "Meta-WebIndexer",
    "operator": "Meta",
    "category": "ai-search"
   },
   {
    "slug": "chatgpt-agent",
    "name": "ChatGPT Agent",
    "operator": "OpenAI",
    "category": "user-fetch"
   },
   {
    "slug": "wpbot",
    "name": "wpbot",
    "operator": "QuantumCloud",
    "category": "tool"
   },
   {
    "slug": "crawl4ai",
    "name": "Crawl4AI",
    "operator": "Crawl4AI project",
    "category": "tool"
   }
  ],
  "blocked_but_does_not_document_obedience": [],
  "tokens_in_your_file_matching_no_known_crawler": [],
  "detail": [
   {
    "slug": "gptbot",
    "name": "GPTBot",
    "operator": "OpenAI",
    "category": "ai-training",
    "robots_token": "GPTBot",
    "named_explicitly": true,
    "matched_group": "GPTBot",
    "allowed": false,
    "decided_by": "Disallow: /",
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your content is excluded from training data for future OpenAI models. No effect on ChatGPT search visibility, on citations, or on links a user pastes into ChatGPT."
   },
   {
    "slug": "oai-searchbot",
    "name": "OAI-SearchBot",
    "operator": "OpenAI",
    "category": "ai-search",
    "robots_token": "OAI-SearchBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "High. Blocking this removes you from ChatGPT search results and from the source links ChatGPT shows. This is the single most expensive block on this list for anyone who wants to be cited by an assistant."
   },
   {
    "slug": "chatgpt-user",
    "name": "ChatGPT-User",
    "operator": "OpenAI",
    "category": "user-fetch",
    "robots_token": "ChatGPT-User",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "ChatGPT cannot open your pages when a user explicitly asks it to. The user sees a fetch failure. This is usually the last bot anyone means to block."
   },
   {
    "slug": "claudebot",
    "name": "ClaudeBot",
    "operator": "Anthropic",
    "category": "ai-training",
    "robots_token": "ClaudeBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Content excluded from training data for future Claude models. No effect on Claude's ability to fetch a link a user gives it."
   },
   {
    "slug": "claude-searchbot",
    "name": "Claude-SearchBot",
    "operator": "Anthropic",
    "category": "ai-search",
    "robots_token": "Claude-SearchBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You stop appearing in Claude's search results and citations."
   },
   {
    "slug": "claude-user",
    "name": "Claude-User",
    "operator": "Anthropic",
    "category": "user-fetch",
    "robots_token": "Claude-User",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Claude reports a fetch failure to a user who asked for your page by name."
   },
   {
    "slug": "anthropic-ai",
    "name": "anthropic-ai",
    "operator": "Anthropic",
    "category": "ai-training",
    "robots_token": "anthropic-ai",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "n-a",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: n-a). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "None. Nothing crawls under this name today; keeping the rule is harmless insurance."
   },
   {
    "slug": "claude-web",
    "name": "Claude-Web",
    "operator": "Anthropic",
    "category": "ai-search",
    "robots_token": "Claude-Web",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "n-a",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: n-a). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "None in practice. Retain the rule; expect no traffic."
   },
   {
    "slug": "google-extended",
    "name": "Google-Extended",
    "operator": "Google",
    "category": "ai-training",
    "robots_token": "Google-Extended",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "n-a",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: n-a). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You are excluded from Gemini grounding and Gemini training. Google Search ranking and indexing are explicitly unaffected. This is the cleanest 'no training, keep my search traffic' lever that exists."
   },
   {
    "slug": "googlebot",
    "name": "Googlebot",
    "operator": "Google",
    "category": "search",
    "robots_token": "Googlebot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Total. You leave Google Search. Never block this to avoid AI use; use Google-Extended instead."
   },
   {
    "slug": "googleother",
    "name": "GoogleOther",
    "operator": "Google",
    "category": "ai-training",
    "robots_token": "GoogleOther",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "No effect on Search indexing. Blocks internal Google research and product fetches."
   },
   {
    "slug": "google-cloudvertexbot",
    "name": "Google-CloudVertexBot",
    "operator": "Google",
    "category": "ai-search",
    "robots_token": "Google-CloudVertexBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Third parties can no longer build Vertex AI agents that read your site. Irrelevant to Google Search."
   },
   {
    "slug": "google-inspectiontool",
    "name": "Google-InspectionTool",
    "operator": "Google",
    "category": "tool",
    "robots_token": "Google-InspectionTool",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your own Search Console live tests stop working. Blocking this only hurts you."
   },
   {
    "slug": "googlebot-image",
    "name": "Googlebot-Image",
    "operator": "Google",
    "category": "search",
    "robots_token": "Googlebot-Image",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your images stop appearing in Google Images."
   },
   {
    "slug": "googlebot-news",
    "name": "Googlebot-News",
    "operator": "Google",
    "category": "search",
    "robots_token": "Googlebot-News",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Google News, with normal Search unaffected."
   },
   {
    "slug": "storebot-google",
    "name": "Storebot-Google",
    "operator": "Google",
    "category": "search",
    "robots_token": "Storebot-Google",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Product listings may lose shopping-specific enrichment. Irrelevant to non-commerce sites."
   },
   {
    "slug": "bingbot",
    "name": "bingbot",
    "operator": "Microsoft",
    "category": "search",
    "robots_token": "bingbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Very high and very wide: Bing, Copilot, DuckDuckGo and several assistants that resell Bing's index all lose you at once. Use nocache/noarchive rather than blocking."
   },
   {
    "slug": "applebot",
    "name": "Applebot",
    "operator": "Apple",
    "category": "search",
    "robots_token": "Applebot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You disappear from Siri, Spotlight and Safari search suggestions across Apple's install base."
   },
   {
    "slug": "applebot-extended",
    "name": "Applebot-Extended",
    "operator": "Apple",
    "category": "ai-training",
    "robots_token": "Applebot-Extended",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "n-a",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: n-a). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Excluded from Apple Intelligence training. Siri, Spotlight and Safari suggestions are unaffected."
   },
   {
    "slug": "perplexitybot",
    "name": "PerplexityBot",
    "operator": "Perplexity",
    "category": "ai-search",
    "robots_token": "PerplexityBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You stop being indexed and cited by Perplexity, and lose the referral clicks its citations produce."
   },
   {
    "slug": "perplexity-user",
    "name": "Perplexity-User",
    "operator": "Perplexity",
    "category": "user-fetch",
    "robots_token": "Perplexity-User",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Not controllable via robots.txt. If you must stop it, verify by the published IP ranges and block at the edge — and accept that users who ask for your page get an error."
   },
   {
    "slug": "ccbot",
    "name": "CCBot",
    "operator": "Common Crawl",
    "category": "dataset",
    "robots_token": "CCBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Future Common Crawl snapshots exclude you, so downstream training sets lose you too — but only going forward. Existing snapshots are permanent and blocking today does not retract them."
   },
   {
    "slug": "bytespider",
    "name": "Bytespider",
    "operator": "ByteDance",
    "category": "ai-training",
    "robots_token": "Bytespider",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "disputed",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: disputed). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Little to lose. If you want it gone, expect to block by user-agent at the edge rather than to ask politely in robots.txt."
   },
   {
    "slug": "tiktokspider",
    "name": "TikTokSpider",
    "operator": "ByteDance",
    "category": "ai-training",
    "robots_token": "TikTokSpider",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "disputed",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: disputed). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Little to lose unless TikTok search referral matters to you."
   },
   {
    "slug": "meta-externalagent",
    "name": "meta-externalagent",
    "operator": "Meta",
    "category": "ai-training",
    "robots_token": "meta-externalagent",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Excluded from Meta AI training. Link previews on Facebook, Instagram and WhatsApp are unaffected — those are a different bot."
   },
   {
    "slug": "meta-externalfetcher",
    "name": "meta-externalfetcher",
    "operator": "Meta",
    "category": "user-fetch",
    "robots_token": "meta-externalfetcher",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Meta AI cannot read pages users hand it."
   },
   {
    "slug": "facebookexternalhit",
    "name": "facebookexternalhit",
    "operator": "Meta",
    "category": "preview",
    "robots_token": "facebookexternalhit",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Severe and usually accidental. Your links share as bare grey boxes with no title, image or description across Facebook, Instagram, Messenger and WhatsApp. Almost nobody means to block this."
   },
   {
    "slug": "facebookbot",
    "name": "FacebookBot",
    "operator": "Meta",
    "category": "ai-training",
    "robots_token": "FacebookBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Negligible today. Keep the rule; expect little traffic."
   },
   {
    "slug": "amazonbot",
    "name": "Amazonbot",
    "operator": "Amazon",
    "category": "ai-search",
    "robots_token": "Amazonbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Alexa and Amazon's assistants stop answering from your pages. Verify with reverse DNS to crawl.amazonbot.amazon before trusting the user-agent."
   },
   {
    "slug": "duckassistbot",
    "name": "DuckAssistBot",
    "operator": "DuckDuckGo",
    "category": "ai-search",
    "robots_token": "DuckAssistBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "No DuckAssist answers or citations from your site. Ordinary DuckDuckGo results are unaffected."
   },
   {
    "slug": "duckduckbot",
    "name": "DuckDuckBot",
    "operator": "DuckDuckGo",
    "category": "search",
    "robots_token": "DuckDuckBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Limited on its own; the real DuckDuckGo lever is bingbot."
   },
   {
    "slug": "ai2bot",
    "name": "AI2Bot",
    "operator": "Allen Institute for AI",
    "category": "dataset",
    "robots_token": "AI2Bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Excluded from open research datasets. Worth a deliberate decision: this is the category where 'blocking AI' also blocks the open, auditable end of it."
   },
   {
    "slug": "ai2bot-dolma",
    "name": "Ai2Bot-Dolma",
    "operator": "Allen Institute for AI",
    "category": "dataset",
    "robots_token": "Ai2Bot-Dolma",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Same as AI2Bot: exclusion from an open, published training corpus."
   },
   {
    "slug": "cohere-ai",
    "name": "cohere-ai",
    "operator": "Cohere",
    "category": "user-fetch",
    "robots_token": "cohere-ai",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Cohere-powered assistants cannot read your pages on request."
   },
   {
    "slug": "cohere-training-data-crawler",
    "name": "cohere-training-data-crawler",
    "operator": "Cohere",
    "category": "ai-training",
    "robots_token": "cohere-training-data-crawler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Excluded from Cohere model training."
   },
   {
    "slug": "mistralai-user",
    "name": "MistralAI-User",
    "operator": "Mistral AI",
    "category": "user-fetch",
    "robots_token": "MistralAI-User",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Le Chat cannot open links your readers give it."
   },
   {
    "slug": "youbot",
    "name": "YouBot",
    "operator": "You.com",
    "category": "ai-search",
    "robots_token": "YouBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from You.com's index and from answers built on its API."
   },
   {
    "slug": "diffbot",
    "name": "Diffbot",
    "operator": "Diffbot",
    "category": "dataset",
    "robots_token": "Diffbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your facts stop entering a widely-licensed knowledge graph. Whether that is a loss depends on whether you want to be a machine-readable entity."
   },
   {
    "slug": "omgilibot",
    "name": "omgilibot",
    "operator": "Webz.io",
    "category": "dataset",
    "robots_token": "omgilibot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Exclusion from a commercial dataset resold to third parties."
   },
   {
    "slug": "omgili",
    "name": "omgili",
    "operator": "Webz.io",
    "category": "dataset",
    "robots_token": "omgili",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Same as omgilibot."
   },
   {
    "slug": "webzio-extended",
    "name": "Webzio-Extended",
    "operator": "Webz.io",
    "category": "ai-training",
    "robots_token": "Webzio-Extended",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your content is excluded from the AI-training tier of Webz.io's product while ordinary collection continues."
   },
   {
    "slug": "imagesiftbot",
    "name": "ImagesiftBot",
    "operator": "Hive AI",
    "category": "dataset",
    "robots_token": "ImagesiftBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your images stop entering an image dataset and reverse-image index."
   },
   {
    "slug": "timpibot",
    "name": "Timpibot",
    "operator": "Timpi",
    "category": "search",
    "robots_token": "Timpibot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Absence from a small independent index."
   },
   {
    "slug": "semrushbot",
    "name": "SemrushBot",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your competitors' Semrush reports get thinner, and so do yours. No user-facing effect."
   },
   {
    "slug": "semrushbot-ocob",
    "name": "SemrushBot-OCOB",
    "operator": "Semrush",
    "category": "ai-training",
    "robots_token": "SemrushBot-OCOB",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Exclusion from Semrush's AI corpus, with its SEO crawl unaffected."
   },
   {
    "slug": "ahrefsbot",
    "name": "AhrefsBot",
    "operator": "Ahrefs",
    "category": "seo",
    "robots_token": "AhrefsBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "No user-facing effect. Ahrefs honours Crawl-delay, so rate-limiting is usually better than blocking."
   },
   {
    "slug": "archive-org-bot",
    "name": "archive.org_bot",
    "operator": "Internet Archive",
    "category": "archive",
    "robots_token": "archive.org_bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your site stops being preserved. When it dies, it is gone. Consider this one separately from the AI question."
   },
   {
    "slug": "ia-archiver",
    "name": "ia_archiver",
    "operator": "Internet Archive",
    "category": "archive",
    "robots_token": "ia_archiver",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Negligible today; retain for tidiness."
   },
   {
    "slug": "yandexbot",
    "name": "YandexBot",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Yandex Search. Verify with reverse DNS to a yandex.ru, yandex.net or yandex.com host — YandexBot is among the most-spoofed user-agents there is."
   },
   {
    "slug": "baiduspider",
    "name": "Baiduspider",
    "operator": "Baidu",
    "category": "search",
    "robots_token": "Baiduspider",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Baidu Search, which matters only if you want Chinese-language traffic."
   },
   {
    "slug": "seznambot",
    "name": "SeznamBot",
    "operator": "Seznam",
    "category": "search",
    "robots_token": "SeznamBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Seznam. Also removes you from its IndexNow endpoint's usefulness."
   },
   {
    "slug": "yeti",
    "name": "Yeti",
    "operator": "Naver",
    "category": "search",
    "robots_token": "Yeti",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Naver, which is most of Korean search."
   },
   {
    "slug": "petalbot",
    "name": "PetalBot",
    "operator": "Huawei",
    "category": "search",
    "robots_token": "PetalBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Petal Search. Frequently blocked for volume rather than for policy."
   },
   {
    "slug": "firecrawlagent",
    "name": "FirecrawlAgent",
    "operator": "Firecrawl",
    "category": "tool",
    "robots_token": "FirecrawlAgent",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Applications built on Firecrawl cannot read your pages. This is increasingly how agents fetch the web, so it is a bigger block than its name suggests."
   },
   {
    "slug": "scrapy",
    "name": "Scrapy",
    "operator": "Scrapy project",
    "category": "tool",
    "robots_token": "Scrapy",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You block a very large tail of unattributed one-off crawlers, and also every well-behaved researcher who did not change the default."
   },
   {
    "slug": "img2dataset",
    "name": "img2dataset",
    "operator": "LAION / img2dataset",
    "category": "dataset",
    "robots_token": "img2dataset",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your images are skipped when someone materialises an image-text dataset that references them."
   },
   {
    "slug": "googlebot-video",
    "name": "Googlebot-Video",
    "operator": "Google",
    "category": "search",
    "robots_token": "Googlebot-Video",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your videos leave Google video search. A rule for Googlebot already covers it, so blocking this token alone is usually a mistake of precision rather than of intent."
   },
   {
    "slug": "googleother-image",
    "name": "GoogleOther-Image",
    "operator": "Google",
    "category": "ai-training",
    "robots_token": "GoogleOther-Image",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Google teams outside Search stop fetching your images. Image Search itself is unaffected — that is Googlebot-Image."
   },
   {
    "slug": "googleother-video",
    "name": "GoogleOther-Video",
    "operator": "Google",
    "category": "ai-training",
    "robots_token": "GoogleOther-Video",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "No effect on Search or on Google Video search. Blocks internal Google research fetches of your video files."
   },
   {
    "slug": "apis-google",
    "name": "APIs-Google",
    "operator": "Google",
    "category": "tool",
    "robots_token": "APIs-Google",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Google API push notifications stop arriving at your endpoint. This only affects services you set up yourself; there is no search or AI consequence."
   },
   {
    "slug": "adsbot-google",
    "name": "AdsBot-Google",
    "operator": "Google",
    "category": "tool",
    "robots_token": "AdsBot-Google",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Google Ads cannot score your landing pages, which lowers Ad Rank on the ads pointing at them. If you do not buy ads, blocking it costs nothing but bandwidth savings."
   },
   {
    "slug": "adsbot-google-mobile",
    "name": "AdsBot-Google-Mobile",
    "operator": "Google",
    "category": "tool",
    "robots_token": "AdsBot-Google-Mobile",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Mobile ad landing pages go unscored and the ads pointing at them rank worse. No effect on organic search."
   },
   {
    "slug": "adsbot-google-mobile-apps",
    "name": "AdsBot-Google-Mobile-Apps",
    "operator": "Google",
    "category": "tool",
    "robots_token": "AdsBot-Google-Mobile-Apps",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "App-install ad landing pages go unscored. Nothing organic changes."
   },
   {
    "slug": "mediapartners-google",
    "name": "Mediapartners-Google",
    "operator": "Google",
    "category": "tool",
    "robots_token": "Mediapartners-Google",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Pages it cannot read get generic, lower-value AdSense ads or none at all. This is the one block on this list that costs you money directly if you run AdSense."
   },
   {
    "slug": "google-safety",
    "name": "Google-Safety",
    "operator": "Google",
    "category": "tool",
    "robots_token": "Google-Safety",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Nothing you can control. The rule is ignored by design; listing the token is documentation, not enforcement."
   },
   {
    "slug": "feedfetcher-google",
    "name": "FeedFetcher-Google",
    "operator": "Google",
    "category": "tool",
    "robots_token": "FeedFetcher-Google",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Feed-driven Google products stop seeing your updates. A robots.txt rule will not stop it — block by user-agent at the edge if you mean it."
   },
   {
    "slug": "google-read-aloud",
    "name": "Google-Read-Aloud",
    "operator": "Google",
    "category": "user-fetch",
    "robots_token": "Google-Read-Aloud",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A reader who asked Google to read your page aloud — often someone using it for accessibility — gets an error instead."
   },
   {
    "slug": "google-site-verification",
    "name": "Google-Site-Verification",
    "operator": "Google",
    "category": "tool",
    "robots_token": "Google-Site-Verification",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your own Search Console verification fails. Blocking this only ever hurts the person doing the blocking."
   },
   {
    "slug": "google-cws",
    "name": "Google-CWS",
    "operator": "Google",
    "category": "tool",
    "robots_token": "Google-CWS",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Chrome Web Store listings that point at your pages cannot fetch them. Relevant only if you publish extensions."
   },
   {
    "slug": "google-pinpoint",
    "name": "Google-Pinpoint",
    "operator": "Google",
    "category": "user-fetch",
    "robots_token": "Google-Pinpoint",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A researcher who explicitly added your page to a collection cannot load it."
   },
   {
    "slug": "googleproducer",
    "name": "GoogleProducer",
    "operator": "Google",
    "category": "tool",
    "robots_token": "GoogleProducer",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your own Google News landing pages stop updating. Only publishers who configured Publisher Center are affected."
   },
   {
    "slug": "googlemessages",
    "name": "GoogleMessages",
    "operator": "Google",
    "category": "preview",
    "robots_token": "GoogleMessages",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your links appear as bare URLs with no title or image in Google Messages chats. A preview fetcher is almost never the one you meant to block."
   },
   {
    "slug": "google-gemininotebook",
    "name": "Google-GeminiNotebook",
    "operator": "Google",
    "category": "user-fetch",
    "robots_token": "Google-GeminiNotebook",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A user who deliberately added your page as a research source gets nothing. This is a citation-shaped fetch, not a training crawl."
   },
   {
    "slug": "google-agent",
    "name": "Google-Agent",
    "operator": "Google",
    "category": "user-fetch",
    "robots_token": "Google-Agent",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Google-hosted agents cannot complete a task on your site for a user who asked them to. This is the agentic-commerce fetch: blocking it removes you from what an assistant can actually do rather than from what it can say."
   },
   {
    "slug": "yandeximages",
    "name": "YandexImages",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexImages",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your images leave Yandex Images, which is a large share of image search in Russian-speaking markets."
   },
   {
    "slug": "yandexvideo",
    "name": "YandexVideo",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexVideo",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Removal from Yandex video search. Note that a second robot, YandexVideoParser, does the same job and is documented as NOT taking the general rules into account."
   },
   {
    "slug": "yandexmedia",
    "name": "YandexMedia",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexMedia",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your multimedia content stops appearing in Yandex's media surfaces."
   },
   {
    "slug": "yandexblogs",
    "name": "YandexBlogs",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexBlogs",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Comment threads and blog posts stop being findable through Yandex blog search."
   },
   {
    "slug": "yandexmarket",
    "name": "YandexMarket",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexMarket",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your products stop being listed and priced in Yandex Market. For a retailer in that market this is a revenue block, not a bandwidth one."
   },
   {
    "slug": "yandexwebmaster",
    "name": "YandexWebmaster",
    "operator": "Yandex",
    "category": "tool",
    "robots_token": "YandexWebmaster",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your own Yandex Webmaster checks stop working. Blocking this only hurts you."
   },
   {
    "slug": "yandexmobilebot",
    "name": "YandexMobileBot",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexMobileBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Yandex loses its mobile-friendliness signal for your pages, which affects how they are ranked and rendered on phones."
   },
   {
    "slug": "yandexfavicons",
    "name": "YandexFavicons",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexFavicons",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your results in Yandex lose their icon. Cosmetic, and a * rule will not achieve it anyway."
   },
   {
    "slug": "yandexcalendar",
    "name": "YandexCalendar",
    "operator": "Yandex",
    "category": "user-fetch",
    "robots_token": "YandexCalendar",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Users who subscribed to a calendar you publish stop receiving updates."
   },
   {
    "slug": "yandexdirect",
    "name": "YandexDirect",
    "operator": "Yandex",
    "category": "tool",
    "robots_token": "YandexDirect",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Ads on your pages become less relevant and earn less. Relevant only if you monetise with Yandex's network."
   },
   {
    "slug": "yandexmetrika",
    "name": "YandexMetrika",
    "operator": "Yandex",
    "category": "tool",
    "robots_token": "YandexMetrika",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Nothing you can enforce through robots.txt. If you run Metrica, its session replay loses your stylesheets and renders your pages wrong."
   },
   {
    "slug": "yandexrenderresourcesbot",
    "name": "YandexRenderResourcesBot",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexRenderResourcesBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Yandex renders your pages without their stylesheets or scripts and ranks what it sees. This is the classic accidental self-inflicted ranking loss."
   },
   {
    "slug": "yandexscreenshotbot",
    "name": "YandexScreenshotBot",
    "operator": "Yandex",
    "category": "tool",
    "robots_token": "YandexScreenshotBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Yandex surfaces that show a page thumbnail show nothing for you."
   },
   {
    "slug": "yandexadditional",
    "name": "YandexAdditional",
    "operator": "Yandex",
    "category": "ai-training",
    "robots_token": "YandexAdditional",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You disappear from Yandex's AI answers while staying in Yandex Search. This is Yandex's equivalent of Google-Extended, and it is the cheap opt-out most people are looking for."
   },
   {
    "slug": "yandexadditionalbot",
    "name": "YandexAdditionalBot",
    "operator": "Yandex",
    "category": "ai-training",
    "robots_token": "YandexAdditionalBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Same as YandexAdditional: out of Yandex's AI answers, still in Yandex Search. Name both tokens or neither."
   },
   {
    "slug": "yandexcombot",
    "name": "YandexComBot",
    "operator": "Yandex",
    "category": "search",
    "robots_token": "YandexComBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "own-token-only",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: own-token-only). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You leave Yandex's non-Russian index. A rule naming this token is the only one that works on it."
   },
   {
    "slug": "siteauditbot",
    "name": "SiteAuditBot",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SiteAuditBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Site Audit reports on your own domain stop working. If a customer is auditing your site with your permission, blocking this breaks their tooling and nothing of yours."
   },
   {
    "slug": "semrushbot-ba",
    "name": "SemrushBot-BA",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot-BA",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "None to you. It costs the site being audited a little accuracy in their backlink report."
   },
   {
    "slug": "semrushbot-si",
    "name": "SemrushBot-SI",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot-SI",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Nothing user-facing. Semrush customers lose on-page suggestions for pages on your domain."
   },
   {
    "slug": "semrushbot-swa",
    "name": "SemrushBot-SWA",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot-SWA",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Writers using Semrush's assistant see your links reported as unreachable."
   },
   {
    "slug": "splitsignalbot",
    "name": "SplitSignalBot",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SplitSignalBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "A site owner's own A/B testing stops. Only relevant on domains whose owner uses the product."
   },
   {
    "slug": "semrushbot-ft",
    "name": "SemrushBot-FT",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot-FT",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your text stops being compared against other people's submissions — which also means copies of your text are less likely to be caught."
   },
   {
    "slug": "semrushbot-esi",
    "name": "SemrushBot-ESI",
    "operator": "Semrush",
    "category": "seo",
    "robots_token": "SemrushBot-ESI",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Enterprise customers lose analysis of your domain. Nothing user-facing."
   },
   {
    "slug": "ahrefssiteaudit",
    "name": "AhrefsSiteAudit",
    "operator": "Ahrefs",
    "category": "seo",
    "robots_token": "AhrefsSiteAudit",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Site owners auditing your domain get an incomplete report. Blocking it saves bandwidth and costs you nothing in search."
   },
   {
    "slug": "mj12bot",
    "name": "MJ12bot",
    "operator": "Majestic",
    "category": "seo",
    "robots_token": "MJ12bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave the Majestic backlink index. No search or AI effect. It supports Crawl-delay, which is usually the better answer than a block."
   },
   {
    "slug": "dotbot",
    "name": "DotBot",
    "operator": "Moz",
    "category": "seo",
    "robots_token": "dotbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave Moz's link index, so Domain Authority and link reports about your site get thinner. Nothing a reader or an assistant sees changes."
   },
   {
    "slug": "rogerbot",
    "name": "rogerbot",
    "operator": "Moz",
    "category": "seo",
    "robots_token": "rogerbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Moz Pro site audits of your domain stop. If the domain is yours and you use Moz, blocking this breaks your own reports."
   },
   {
    "slug": "dataforseobot",
    "name": "DataForSeoBot",
    "operator": "DataForSEO",
    "category": "seo",
    "robots_token": "DataForSeoBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave a dataset that a long tail of SEO products is built on. No user-facing effect."
   },
   {
    "slug": "serpstatbot",
    "name": "serpstatbot",
    "operator": "Serpstat",
    "category": "seo",
    "robots_token": "serpstatbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave Serpstat's link index. Try Crawl-delay first — they honour it, and a slow crawler is cheaper to keep than to fight."
   },
   {
    "slug": "barkrowler",
    "name": "Barkrowler",
    "operator": "Babbar",
    "category": "seo",
    "robots_token": "barkrowler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave Babbar's index. No effect on search or assistants."
   },
   {
    "slug": "screaming-frog-seo-spider",
    "name": "Screaming Frog SEO Spider",
    "operator": "Screaming Frog",
    "category": "tool",
    "robots_token": "Screaming Frog SEO Spider",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You block a consultant auditing your own site as often as you block a stranger. Treat it as a rate-limit question, not a consent one — and note that the user-agent is configurable, so a block is advisory."
   },
   {
    "slug": "seokicks",
    "name": "SEOkicks",
    "operator": "SEOkicks",
    "category": "seo",
    "robots_token": "SEOkicks",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave a regional backlink index. Nothing else changes."
   },
   {
    "slug": "mojeekbot",
    "name": "MojeekBot",
    "operator": "Mojeek",
    "category": "search",
    "robots_token": "MojeekBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave an independent index that other privacy-focused search products draw on. Small traffic, disproportionate long-term cost to web plurality."
   },
   {
    "slug": "kagibot",
    "name": "Kagibot",
    "operator": "Kagi",
    "category": "search",
    "robots_token": "Kagibot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave Kagi's index. Kagi's users are paying to search and skew technical; per visitor this is an expensive block."
   },
   {
    "slug": "qwantbot",
    "name": "Qwantbot",
    "operator": "Qwant",
    "category": "search",
    "robots_token": "Qwantbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You leave the index behind Qwant, the French privacy-focused engine, and the products that federate it."
   },
   {
    "slug": "qwantbot-news",
    "name": "Qwantbot-news",
    "operator": "Qwant",
    "category": "search",
    "robots_token": "Qwantbot-news",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your articles stop appearing in Qwant News. A rule for Qwantbot as a substring already catches both."
   },
   {
    "slug": "slackbot-linkexpanding",
    "name": "Slackbot-LinkExpanding",
    "operator": "Slack",
    "category": "preview",
    "robots_token": "Slackbot-LinkExpanding",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your links appear in Slack as bare URLs. Inside working teams that quietly costs you clicks, and nothing is gained."
   },
   {
    "slug": "slackbot",
    "name": "Slackbot",
    "operator": "Slack",
    "category": "preview",
    "robots_token": "Slackbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Slack stops being able to read your robots.txt, which is a strange thing to want. Block the link expander instead if that is the goal."
   },
   {
    "slug": "pinterestbot",
    "name": "Pinterestbot",
    "operator": "Pinterest",
    "category": "search",
    "robots_token": "Pinterestbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Pins pointing at your site go stale — wrong prices, dead links — and new content stops being indexed. For a retailer this is one of the more expensive blocks on the list."
   },
   {
    "slug": "bedrockbot",
    "name": "bedrockbot",
    "operator": "Amazon",
    "category": "ai-search",
    "robots_token": "bedrockbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Companies building retrieval applications on Bedrock cannot include your pages. This is a RAG block, not a training block: nothing is being trained, but nothing can cite you either."
   },
   {
    "slug": "cloudflare-autorag",
    "name": "Cloudflare-AutoRAG",
    "operator": "Cloudflare",
    "category": "ai-search",
    "robots_token": "Cloudflare-AutoRAG",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Applications built on Cloudflare AI Search cannot retrieve your pages. If you are the one building the index over your own site, blocking it breaks your own product."
   },
   {
    "slug": "exasearchbot",
    "name": "ExaSearchBot",
    "operator": "Exa",
    "category": "ai-search",
    "robots_token": "ExaSearchBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Agents built on Exa's API stop finding you. Exa publishes no statement about robots.txt compliance, so treat the rule as a request."
   },
   {
    "slug": "shapbot",
    "name": "ShapBot",
    "operator": "Parallel",
    "category": "ai-search",
    "robots_token": "ShapBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Agents using Parallel's research API lose you as a source. Parallel documents robots.txt compliance, so a rule works."
   },
   {
    "slug": "terracotta",
    "name": "TerraCotta",
    "operator": "Ceramic AI",
    "category": "ai-search",
    "robots_token": "TerraCotta",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You are absent from another agent-facing retrieval index. Ceramic documents that it obeys robots.txt."
   },
   {
    "slug": "crawlspace",
    "name": "Crawlspace",
    "operator": "Crawlspace",
    "category": "tool",
    "robots_token": "Crawlspace",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Whatever any Crawlspace customer was building over your pages stops working. Volume and intent vary per customer, so this is a rate-limit decision more than a consent one."
   },
   {
    "slug": "panscient",
    "name": "Panscient",
    "operator": "Panscient",
    "category": "dataset",
    "robots_token": "panscient.com",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your company pages stop feeding a business-data product. No effect on search or assistants."
   },
   {
    "slug": "sbintuitionsbot",
    "name": "SBIntuitionsBot",
    "operator": "SB Intuitions",
    "category": "ai-training",
    "robots_token": "SBIntuitionsBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your content is excluded from a Japanese-language foundation-model corpus. Nothing user-facing changes."
   },
   {
    "slug": "icc-crawler",
    "name": "ICC-Crawler",
    "operator": "NICT",
    "category": "ai-training",
    "robots_token": "ICC-Crawler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "You are excluded from a national research corpus and from the commercial redistributions of it. This is a dataset-shaped block: one refusal, many downstream effects."
   },
   {
    "slug": "cotoyogi",
    "name": "Cotoyogi",
    "operator": "ROIS-DS",
    "category": "ai-training",
    "robots_token": "Cotoyogi",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your Japanese-language content is left out of an academic training corpus."
   },
   {
    "slug": "isscyberriskcrawler",
    "name": "ISSCyberRiskCrawler",
    "operator": "ISS Corporate Solutions",
    "category": "ai-training",
    "robots_token": "ISSCyberRiskCrawler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "disputed",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: disputed). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A rule here is a statement of intent. Your organisation's public footprint still gets scored — by a model trained on everybody else."
   },
   {
    "slug": "sidetrade-indexer-bot",
    "name": "Sidetrade indexer bot",
    "operator": "Sidetrade",
    "category": "ai-training",
    "robots_token": "Sidetrade indexer bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Exclusion from a commercial B2B dataset. The operator publishes no robots.txt statement, so treat the rule as a request rather than a control."
   },
   {
    "slug": "yak",
    "name": "YaK",
    "operator": "Meltwater",
    "category": "dataset",
    "robots_token": "YaK",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your content stops appearing in Meltwater's media monitoring — which is how PR teams find out you were mentioned. Some publishers want to be in it."
   },
   {
    "slug": "atlassian-bot",
    "name": "atlassian-bot",
    "operator": "Atlassian",
    "category": "ai-search",
    "robots_token": "atlassian-bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Rovo cannot answer from your public documentation. If your customers live inside Atlassian tools, this is a support-deflection block."
   },
   {
    "slug": "klaviyoaibot",
    "name": "KlaviyoAIBot",
    "operator": "Klaviyo",
    "category": "ai-search",
    "robots_token": "KlaviyoAIBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "If the connected domain is yours, blocking this breaks the agent you configured. If it is not, this bot should not be reaching you at all."
   },
   {
    "slug": "quillbot",
    "name": "QuillBot",
    "operator": "QuillBot",
    "category": "ai-training",
    "robots_token": "QuillBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Exclusion from QuillBot's corpus. No compliance statement is published, so the rule is a request."
   },
   {
    "slug": "phindbot",
    "name": "PhindBot",
    "operator": "Phind",
    "category": "ai-search",
    "robots_token": "PhindBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You stop being cited in answers to technical questions — which, for documentation and reference sites, is the exact audience most worth keeping."
   },
   {
    "slug": "andibot",
    "name": "Andibot",
    "operator": "Andi",
    "category": "ai-search",
    "robots_token": "Andibot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You disappear from another assistant's answers. Andi publishes no robots.txt statement."
   },
   {
    "slug": "anomura",
    "name": "Anomura",
    "operator": "Direqt",
    "category": "ai-search",
    "robots_token": "Anomura",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "If you are the publisher, this breaks the assistant you put on your own pages. If you are not, it should not be crawling you."
   },
   {
    "slug": "aiwebindex",
    "name": "AIWebIndex",
    "operator": "Lyrenth",
    "category": "ai-search",
    "robots_token": "AIWebIndex",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Agents reading through this index stop seeing you — including the attribution and link back that make it a referral rather than a summary."
   },
   {
    "slug": "factset-spyderbot",
    "name": "Factset_spyderbot",
    "operator": "FactSet",
    "category": "ai-training",
    "robots_token": "Factset_spyderbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Exclusion from a financial-data vendor's corpus. Relevant mostly to companies whose filings and disclosures are being read."
   },
   {
    "slug": "poseidon-research-crawler",
    "name": "Poseidon Research Crawler",
    "operator": "Poseidon Research",
    "category": "ai-training",
    "robots_token": "Poseidon Research Crawler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Exclusion from an interpretability research corpus. No published compliance statement."
   },
   {
    "slug": "qualifiedbot",
    "name": "QualifiedBot",
    "operator": "Qualified",
    "category": "ai-search",
    "robots_token": "QualifiedBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A chatbot on a site that licensed the product loses context. If that site is yours, this block is self-inflicted."
   },
   {
    "slug": "reflectionbot",
    "name": "Reflectionbot",
    "operator": "Reflection AI",
    "category": "ai-training",
    "robots_token": "Reflectionbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Unknown by construction — which is itself the reason some people block it. Nothing user-facing depends on it."
   },
   {
    "slug": "thinkbot",
    "name": "Thinkbot",
    "operator": "Thinkbot",
    "category": "dataset",
    "robots_token": "Thinkbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "disputed",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: disputed). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Exclusion from a market-research dataset. Expect to enforce this at the edge rather than in robots.txt."
   },
   {
    "slug": "aihitbot",
    "name": "aiHitBot",
    "operator": "aiHit",
    "category": "dataset",
    "robots_token": "aiHitBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your company record in a B2B dataset goes stale. Documented as respecting robots.txt, so the rule works."
   },
   {
    "slug": "linguee-bot",
    "name": "Linguee Bot",
    "operator": "Linguee",
    "category": "ai-training",
    "robots_token": "Linguee Bot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "disputed",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: disputed). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Multilingual pages stop feeding a translation corpus. If your site is translated, being in it is usually a benefit."
   },
   {
    "slug": "lightpanda",
    "name": "Lightpanda",
    "operator": "Lightpanda",
    "category": "tool",
    "robots_token": "Lightpanda",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You block a browser, not a company: the same rule stops a scraper and a legitimate automation a customer of yours is running."
   },
   {
    "slug": "laiondownloader",
    "name": "LAIONDownloader",
    "operator": "LAION / img2dataset",
    "category": "dataset",
    "robots_token": "LAIONDownloader",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "by-design-no",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: by-design-no). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Your media is skipped when an open research dataset is built from URL lists. Once a dataset is published, a later block does not remove you from it."
   },
   {
    "slug": "velenpublicwebcrawler",
    "name": "VelenPublicWebCrawler",
    "operator": "Hunter (Velen)",
    "category": "dataset",
    "robots_token": "VelenPublicWebCrawler",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your company pages stop feeding a B2B contact and company dataset. The crawl rate it documents makes this one of the cheapest visitors to simply allow."
   },
   {
    "slug": "awariosmartbot",
    "name": "AwarioSmartBot",
    "operator": "Awario",
    "category": "dataset",
    "robots_token": "AwarioSmartBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Mentions of brands on your pages stop being surfaced to the people monitoring them — including, quite possibly, your own."
   },
   {
    "slug": "awariorssbot",
    "name": "AwarioRssBot",
    "operator": "Awario",
    "category": "dataset",
    "robots_token": "AwarioRssBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Your RSS updates stop reaching Awario's monitoring. Block both tokens or neither."
   },
   {
    "slug": "echoboxbot",
    "name": "EchoboxBot",
    "operator": "Echobox",
    "category": "dataset",
    "robots_token": "EchoboxBot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "Publishers using Echobox get worse scheduling decisions about your articles. No compliance statement is published."
   },
   {
    "slug": "meta-webindexer",
    "name": "Meta-WebIndexer",
    "operator": "Meta",
    "category": "ai-search",
    "robots_token": "Meta-WebIndexer",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You leave the index Meta AI answers from across Facebook, Instagram and WhatsApp — the largest assistant install base there is. A robots.txt that names the two older Meta tokens does not cover this one."
   },
   {
    "slug": "chatgpt-agent",
    "name": "ChatGPT Agent",
    "operator": "OpenAI",
    "category": "user-fetch",
    "robots_token": "ChatGPT-User",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "documented",
    "enforcement_note": null,
    "what_blocking_costs": "Agentic tasks a user asked for — booking, comparing, filling a form on your site — fail. This is the fetch that ends in a transaction, so it is the most expensive user-triggered block on this list."
   },
   {
    "slug": "wpbot",
    "name": "wpbot",
    "operator": "QuantumCloud",
    "category": "tool",
    "robots_token": "wpbot",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "A WordPress site running that plugin loses its own content as an answer source. Only relevant where the plugin is installed."
   },
   {
    "slug": "crawl4ai",
    "name": "Crawl4AI",
    "operator": "Crawl4AI project",
    "category": "tool",
    "robots_token": "Crawl4AI",
    "named_explicitly": false,
    "matched_group": null,
    "allowed": true,
    "decided_by": null,
    "respects_robots_txt": "undocumented",
    "enforcement_note": "This crawler does not document obedience to robots.txt (index value: undocumented). A robots rule is a request to it; enforce at the edge by IP or user-agent if it matters.",
    "what_blocking_costs": "You block a library, not an operator: the rule catches a researcher and a bulk scraper equally, and anyone who edits one config line is not caught at all."
   }
  ],
  "index": {
   "crawlers_indexed": 150,
   "generated_at": "2026-09-02T05:58:58+00:00",
   "source": "https://www.pathwren.workers.dev/data/agents.json"
  },
  "caveats": [
   "A token this index does not know is not necessarily wrong — it may be a crawler we have not indexed. It is worth checking the spelling against the operator's own documentation.",
   "'Blocked' means a compliant crawler would not fetch this path. Crawlers whose respects_robots_txt is not 'documented' have been observed ignoring it; for those, robots.txt is advisory and the enforcement point is the edge."
  ],
  "source": "https://www.pathwren.workers.dev/mcp/robots"
 },
 "answered_by": {
  "note": "The MCP tool itself answered this. One implementation, two doors — this endpoint holds no copy of it.",
  "server": "robots-policy-lint",
  "endpoint": "https://www.pathwren.workers.dev/mcp/robots",
  "tool": "audit_ai_access",
  "protocol": "MCP streamable-http, JSON-RPC 2.0 tools/call"
 },
 "caveats": [
  "The verdict is computed against /data/agents.json, the same file this host publishes and the same one the MCP tool reads."
 ],
 "docs": "https://www.pathwren.workers.dev/tools/ai-access.html?s=client-dossiers",
 "catalogue": "https://www.pathwren.workers.dev/tools/index.json?s=client-dossiers",
 "more": {
  "who actually crawls this host": "https://www.pathwren.workers.dev/bot/index.html?s=client-dossiers",
  "every named client as one file": "https://www.pathwren.workers.dev/data/observed-clients.json?s=client-dossiers"
 },
 "auth": "none — no account, no key, no handshake. CORS open, cacheable, CC0-1.0.",
 "license": "CC0-1.0",
 "independent": true
}
