{
 "slug": "block-all-ai",
 "title": "Block every AI crawler",
 "summary": "Training, AI search, user-triggered fetches and corpus builders, all refused. Classic search engines still allowed.",
 "detail": "The maximal AI opt-out that still leaves you in Google and Bing. Understand the price before deploying it: you will not be cited by any assistant, and when a reader explicitly asks ChatGPT or Claude to open your page, they get an error. Note also that Perplexity-User and Bytespider are listed here but documented as not governed by robots.txt, so this file is a statement of intent for those two, not an enforcement mechanism.",
 "crawler_count": 35,
 "crawlers": [
  "ai2bot",
  "ai2bot-dolma",
  "amazonbot",
  "anthropic-ai",
  "applebot-extended",
  "bytespider",
  "ccbot",
  "chatgpt-user",
  "claude-searchbot",
  "claude-user",
  "claude-web",
  "claudebot",
  "cohere-ai",
  "cohere-training-data-crawler",
  "diffbot",
  "duckassistbot",
  "facebookbot",
  "google-cloudvertexbot",
  "google-extended",
  "googleother",
  "gptbot",
  "imagesiftbot",
  "img2dataset",
  "meta-externalagent",
  "meta-externalfetcher",
  "mistralai-user",
  "oai-searchbot",
  "omgili",
  "omgilibot",
  "perplexity-user",
  "perplexitybot",
  "semrushbot-ocob",
  "tiktokspider",
  "webzio-extended",
  "youbot"
 ],
 "robots_txt_url": "https://www.pathwren.workers.dev/robots/block-all-ai.txt",
 "robots_txt": "# AI Crawler Index — policy: block-all-ai\n# Block every AI crawler\n# Training, AI search, user-triggered fetches and corpus builders, all refused. Classic search engines still allowed.\n# Generated 2026-09-01 from https://www.pathwren.workers.dev/policy/block-all-ai.html\n# 35 crawlers named. Paste into robots.txt at your document root.\n\nUser-agent: AI2Bot\nDisallow: /\n\nUser-agent: Ai2Bot-Dolma\nDisallow: /\n\nUser-agent: Amazonbot\nDisallow: /\n\nUser-agent: anthropic-ai   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Applebot-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: Bytespider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: CCBot\nDisallow: /\n\nUser-agent: ChatGPT-User\nDisallow: /\n\nUser-agent: Claude-SearchBot\nDisallow: /\n\nUser-agent: Claude-User\nDisallow: /\n\nUser-agent: Claude-Web   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: ClaudeBot\nDisallow: /\n\nUser-agent: cohere-ai\nDisallow: /\n\nUser-agent: cohere-training-data-crawler\nDisallow: /\n\nUser-agent: Diffbot\nDisallow: /\n\nUser-agent: DuckAssistBot\nDisallow: /\n\nUser-agent: FacebookBot\nDisallow: /\n\nUser-agent: Google-CloudVertexBot\nDisallow: /\n\nUser-agent: Google-Extended   # control token, no crawler uses this user-agent\nDisallow: /\n\nUser-agent: GoogleOther\nDisallow: /\n\nUser-agent: GPTBot\nDisallow: /\n\nUser-agent: ImagesiftBot\nDisallow: /\n\nUser-agent: img2dataset\nDisallow: /\n\nUser-agent: meta-externalagent\nDisallow: /\n\nUser-agent: meta-externalfetcher\nDisallow: /\n\nUser-agent: MistralAI-User\nDisallow: /\n\nUser-agent: OAI-SearchBot\nDisallow: /\n\nUser-agent: omgili\nDisallow: /\n\nUser-agent: omgilibot\nDisallow: /\n\nUser-agent: Perplexity-User   # operator states robots.txt does not apply; enforce at the edge\nDisallow: /\n\nUser-agent: PerplexityBot\nDisallow: /\n\nUser-agent: SemrushBot-OCOB\nDisallow: /\n\nUser-agent: TikTokSpider   # compliance disputed; enforce at the edge\nDisallow: /\n\nUser-agent: Webzio-Extended\nDisallow: /\n\nUser-agent: YouBot\nDisallow: /\n\nUser-agent: *\nAllow: /\n\nSitemap: https://www.pathwren.workers.dev/sitemap.xml\n",
 "generated": "2026-09-01"
}