{
 "operator": "Google",
 "slug": "google",
 "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
 "crawler_count": 8,
 "crawlers": [
  {
   "slug": "google-cloudvertexbot",
   "name": "Google-CloudVertexBot",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-search",
   "category_label": "AI search crawlers",
   "robots_token": "Google-CloudVertexBot",
   "user_agent_substring": "Google-CloudVertexBot",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-CloudVertexBot/1.0; +https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 135,
   "ipv6_prefix_count": 135,
   "what_it_is": "Crawls a site on behalf of a Vertex AI Agent Builder customer who is building an agent over that site. It only visits sites the customer has asked it to.",
   "cost_of_blocking": "Third parties can no longer build Vertex AI agents that read your site. Irrelevant to Google Search.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-cloudvertexbot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-cloudvertexbot.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "google-extended",
   "name": "Google-Extended",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Google-Extended",
   "user_agent_substring": "(control token only — no crawler)",
   "user_agent_example": "(none: Google-Extended never appears as a user-agent)",
   "respects_robots_txt": "n-a",
   "respects_robots_txt_label": "control token only — no crawler",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Not a crawler. A robots.txt token that tells Google whether pages Googlebot already fetched may be used to train and ground Gemini. You will never see it in an access log; disallowing it changes what Google does with content it fetched under a different name.",
   "cost_of_blocking": "You are excluded from Gemini grounding and Gemini training. Google Search ranking and indexing are explicitly unaffected. This is the cleanest 'no training, keep my search traffic' lever that exists.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-extended.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-extended.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "google-inspectiontool",
   "name": "Google-InspectionTool",
   "operator": "Google",
   "operator_slug": "google",
   "category": "tool",
   "category_label": "Tools and frameworks",
   "robots_token": "Google-InspectionTool",
   "user_agent_substring": "Google-InspectionTool",
   "user_agent_example": "Mozilla/5.0 (compatible; Google-InspectionTool/1.0;)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 135,
   "ipv6_prefix_count": 135,
   "what_it_is": "The fetcher behind Search Console's URL Inspection and the Rich Results Test. It runs when a site owner clicks a button.",
   "cost_of_blocking": "Your own Search Console live tests stop working. Blocking this only hurts you.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/google-inspectiontool.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/google-inspectiontool.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "googlebot",
   "name": "Googlebot",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot",
   "user_agent_substring": "Googlebot",
   "user_agent_example": "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 169,
   "ipv6_prefix_count": 146,
   "what_it_is": "The classic search crawler. It is also the crawler behind AI Overviews: Google does not run a separate bot for them, which is why the only AI opt-out is the Google-Extended token and not a Googlebot block.",
   "cost_of_blocking": "Total. You leave Google Search. Never block this to avoid AI use; use Google-Extended instead.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "googlebot-image",
   "name": "Googlebot-Image",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot-Image",
   "user_agent_substring": "Googlebot-Image",
   "user_agent_example": "Googlebot-Image/1.0",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 169,
   "ipv6_prefix_count": 146,
   "what_it_is": "Image indexing for Google Images. A separate token so you can leave images out of search without leaving search.",
   "cost_of_blocking": "Your images stop appearing in Google Images.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot-image.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot-image.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "googlebot-news",
   "name": "Googlebot-News",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Googlebot-News",
   "user_agent_substring": "Googlebot-News",
   "user_agent_example": "(uses the Googlebot user-agent; controlled by the Googlebot-News robots token)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json",
   "ipv4_prefix_count": 169,
   "ipv6_prefix_count": 146,
   "what_it_is": "A robots.txt token controlling inclusion in Google News. It does not have its own user-agent string; the fetch arrives as Googlebot.",
   "cost_of_blocking": "Removal from Google News, with normal Search unaffected.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googlebot-news.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googlebot-news.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "googleother",
   "name": "GoogleOther",
   "operator": "Google",
   "operator_slug": "google",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "GoogleOther",
   "user_agent_substring": "GoogleOther",
   "user_agent_example": "Mozilla/5.0 (compatible; GoogleOther)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 135,
   "ipv6_prefix_count": 135,
   "what_it_is": "A generic fetcher used by Google product teams for one-off crawls and research, including data collection that does not belong to Search.",
   "cost_of_blocking": "No effect on Search indexing. Blocks internal Google research and product fetches.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/googleother.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/googleother.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "storebot-google",
   "name": "Storebot-Google",
   "operator": "Google",
   "operator_slug": "google",
   "category": "search",
   "category_label": "Search engines",
   "robots_token": "Storebot-Google",
   "user_agent_substring": "Storebot-Google",
   "user_agent_example": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Safari/537.36 (compatible; Storebot-Google/1.0; +https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "published-ranges",
   "verification_label": "published IP ranges",
   "published_ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "ip_ranges_endpoint": "https://www.pathwren.workers.dev/ip-ranges/google-special.json",
   "ipv4_prefix_count": 135,
   "ipv6_prefix_count": 135,
   "what_it_is": "Checks shopping and checkout flows for Google's shopping surfaces.",
   "cost_of_blocking": "Product listings may lose shopping-specific enrichment. Irrelevant to non-commerce sites.",
   "operator_docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "html_url": "https://www.pathwren.workers.dev/crawler/storebot-google.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/storebot-google.json",
   "last_reviewed": "2026-09-01"
  }
 ],
 "ip_range_endpoints": [
  {
   "slug": "google-googlebot",
   "label": "Googlebot",
   "url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "mirror": "https://www.pathwren.workers.dev/ip-ranges/google-googlebot.json"
  },
  {
   "slug": "google-special",
   "label": "Google special-purpose crawlers",
   "url": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "mirror": "https://www.pathwren.workers.dev/ip-ranges/google-special.json"
  },
  {
   "slug": "google-user-triggered",
   "label": "Google user-triggered fetchers",
   "url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "mirror": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered.json"
  },
  {
   "slug": "google-user-triggered-google",
   "label": "Google user-triggered fetchers (Google-owned ranges)",
   "url": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers-google.json",
   "mirror": "https://www.pathwren.workers.dev/ip-ranges/google-user-triggered-google.json"
  }
 ]
}