{
 "tool": "classify-ua",
 "endpoint": "https://www.pathwren.workers.dev/tools/classify-ua",
 "asked": {
  "ua": "GPTBot/1.2"
 },
 "summary": "GPTBot — OpenAI (AI training crawlers).\nMatched on \"GPTBot\".\n\nWhat it is: OpenAI's bulk crawler. Pages it fetches may be used to train future OpenAI foundation models. It is not the bot that puts you in ChatGPT's search results, and blocking it does not remove you from them.\n\nCost of blocking it: Your content is excluded from training data for future OpenAI models. No effect on ChatGPT search visibility, on citations, or on links a user pastes into ChatGPT.\n\nrobots.txt token: GPTBot · robots.txt: obeys robots.txt (documented) · verification: published IP ranges\n\nA user-agent string is a claim, not a proof: anyone can send it. Before you act on a match, check the address with is_verified_crawler_ip, or use the operator's reverse-DNS method where there are no published ranges.",
 "answer": {
  "user_agent": "GPTBot/1.2",
  "matched": true,
  "matched_on": "GPTBot",
  "crawler": {
   "slug": "gptbot",
   "name": "GPTBot",
   "operator": "OpenAI",
   "operator_slug": "openai",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GPTBot",
   "user_agent_substring": "GPTBot",
   "user_agent_example": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.2; +https://openai.com/gptbot",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://openai.com/gptbot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/openai-gptbot.json",
   "ipv4_prefix_count": 21,
   "ipv6_prefix_count": 0,
   "what_it_is": "OpenAI's bulk crawler. Pages it fetches may be used to train future OpenAI foundation models. It is not the bot that puts you in ChatGPT's search results, and blocking it does not remove you from them.",
   "cost_of_blocking": "Your content is excluded from training data for future OpenAI models. No effect on ChatGPT search visibility, on citations, or on links a user pastes into ChatGPT.",
   "operator_docs": "https://platform.openai.com/docs/bots",
   "docs_url": "https://www.pathwren.workers.dev/crawler/gptbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/gptbot.json",
   "last_reviewed": "2026-09-06"
  },
  "also_matched": [],
  "verification": {
   "method": "published-ranges",
   "label": "published IP ranges",
   "published_ip_ranges": "https://www.pathwren.workers.dev/ip-ranges/openai-gptbot.json",
   "how": "Call is_verified_crawler_ip with the client address; the operator publishes its prefixes."
  },
  "caveat": "A user-agent string is a claim, not a proof: anyone can send it. Before you act on a match, check the address with is_verified_crawler_ip, or use the operator's reverse-DNS method where there are no published ranges.",
  "source": "https://www.pathwren.workers.dev"
 },
 "answered_by": {
  "note": "The MCP tool itself answered this. One implementation, two doors — this endpoint holds no copy of it.",
  "server": "ai-crawler-index",
  "endpoint": "https://www.pathwren.workers.dev/mcp",
  "tool": "classify_user_agent",
  "protocol": "MCP streamable-http, JSON-RPC 2.0 tools/call"
 },
 "caveats": [
  "A user-agent is a claim, not proof. Pair it with /tools/verify-crawler, which is the address half of the same question."
 ],
 "docs": "https://www.pathwren.workers.dev/tools/classify-ua.html?s=client-dossiers",
 "catalogue": "https://www.pathwren.workers.dev/tools/index.json?s=client-dossiers",
 "more": {
  "who actually crawls this host": "https://www.pathwren.workers.dev/bot/index.html?s=client-dossiers",
  "every named client as one file": "https://www.pathwren.workers.dev/data/observed-clients.json?s=client-dossiers"
 },
 "auth": "none — no account, no key, no handshake. CORS open, cacheable, CC0-1.0.",
 "license": "CC0-1.0",
 "independent": true,
 "_about": {
  "page": "https://www.pathwren.workers.dev/bot/claudebot.html?s=client-dossiers",
  "what": "What this host recorded about the client that made this request — the requests, the status codes and the user-agent string, from its own log. Nothing else: no address is published, nothing is looked up anywhere, and there is no account to close.",
  "matched_on": "the User-Agent header you sent. If we have no page for it yet this points at the index of all of them.",
  "index": "https://www.pathwren.workers.dev/bot/index.html?s=client-dossiers"
 }
}
