{
 "feed": "ai-crawler-index changes",
 "self": "https://www.pathwren.workers.dev/changes.json",
 "what": "Everything that changed in this index since your cursor: operator IP-range lists that moved, upstreams that failed or recovered, and crawler records that were added or edited. Read `cursor` from every response and send it back as `since` next time.",
 "cursor": 100,
 "head": 105,
 "since": null,
 "since_kind": "absent",
 "up_to_date": false,
 "count": 100,
 "limit": 100,
 "has_more": true,
 "next": "https://www.pathwren.workers.dev/changes.json?since=100",
 "oldest_cursor": 1,
 "cursor_expired": false,
 "changes": [
  {
   "seq": 1,
   "at": "2026-09-01T14:39:48+00:00",
   "kind": "baseline",
   "title": "Ledger opened: 12 operator IP-range sources, 56 crawler records",
   "detail": "This is the first entry. Every event after it is a real observed difference between two refreshes; nothing before it was recorded, and none is invented. Fetch the documents once at this cursor and poll from here.",
   "sources": [
    "apple-applebot",
    "bing-bingbot",
    "duckduckgo-duckduckbot",
    "google-googlebot",
    "google-special",
    "google-user-triggered",
    "google-user-triggered-google",
    "openai-chatgpt-user",
    "openai-gptbot",
    "openai-searchbot",
    "perplexity-bot",
    "perplexity-user"
   ],
   "sources_failing": [],
   "ipv4_prefixes": 1887,
   "ipv6_prefixes": 1056,
   "crawlers": 56,
   "affects": [
    "/ip-ranges/all.json",
    "/data/agents.json",
    "/status.json"
   ]
  },
  {
   "seq": 2,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "ip_ranges_changed",
   "source": "duckduckgo-duckduckbot",
   "operator": "duckduckgo",
   "label": "DuckDuckBot",
   "upstream": "https://duckduckgo.com/duckduckbot.json",
   "covers": [
    "duckduckbot"
   ],
   "operator_published_at": "2026-09-01T12:44:58.000000",
   "sha256": "fd9039362dfaaad236d50c07ddd12bf0cc7f0d530fbff59b8feb1b22e0db2459",
   "previous_sha256": "48937109ec4302510dc989139e54c7e100f6f30e59ccf8905cb6c2f3370e91f4",
   "ipv4_before": 481,
   "ipv4_after": 486,
   "ipv6_before": 0,
   "ipv6_after": 0,
   "added_count": 5,
   "removed_count": 0,
   "ipv4_added": [
    "135.171.251.169/32",
    "172.188.250.220/32",
    "4.144.148.23/32",
    "4.144.227.142/32",
    "4.144.251.39/32"
   ],
   "ipv4_removed": [],
   "ipv6_added": [],
   "ipv6_removed": [],
   "lists_truncated": false,
   "title": "DuckDuckBot: 5 prefixes added, 0 removed",
   "detail": "The operator's own published list changed and this mirror followed it. Counts are exact; the lists are complete unless lists_truncated is true, in which case re-read /ip-ranges/duckduckgo-duckduckbot.json.",
   "affects": [
    "/ip-ranges/duckduckgo-duckduckbot.json",
    "/ip-ranges/all.json",
    "/ip-ranges/all.txt",
    "/status.json"
   ]
  },
  {
   "seq": 3,
   "at": "2026-09-01T20:40:08+00:00",
   "kind": "ip_ranges_changed",
   "source": "google-googlebot",
   "operator": "google",
   "label": "Googlebot",
   "upstream": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "covers": [
    "googlebot",
    "googlebot-image",
    "googlebot-news"
   ],
   "operator_published_at": "2026-09-01T14:45:54.000000",
   "sha256": "83a097b1fd4e6e9c0a485e162992ea3bf1179c63abe137d1b25c887efd914a23",
   "previous_sha256": "fae5f5822573f048ec859bd3a0073d7887f00f5612ddf0b2f1df1cb168f1bff0",
   "ipv4_before": 169,
   "ipv4_after": 170,
   "ipv6_before": 146,
   "ipv6_after": 147,
   "added_count": 2,
   "removed_count": 0,
   "ipv4_added": [
    "66.249.68.224/27"
   ],
   "ipv4_removed": [],
   "ipv6_added": [
    "2001:4860:4801:43::/64"
   ],
   "ipv6_removed": [],
   "lists_truncated": false,
   "title": "Googlebot: 2 prefixes added, 0 removed",
   "detail": "The operator's own published list changed and this mirror followed it. Counts are exact; the lists are complete unless lists_truncated is true, in which case re-read /ip-ranges/google-googlebot.json.",
   "affects": [
    "/ip-ranges/google-googlebot.json",
    "/ip-ranges/all.json",
    "/ip-ranges/all.txt",
    "/status.json"
   ]
  },
  {
   "seq": 4,
   "at": "2026-09-01T20:40:08+00:00",
   "kind": "ip_ranges_changed",
   "source": "google-special",
   "operator": "google",
   "label": "Google special-purpose crawlers",
   "upstream": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "covers": [
    "googleother",
    "google-cloudvertexbot",
    "google-inspectiontool",
    "storebot-google"
   ],
   "operator_published_at": "2026-09-01T14:46:55.000000",
   "sha256": "df153b01b2d2777cc9b65008b73eca1a8c51063b2903da05a63b1c8a45998f4d",
   "previous_sha256": "0b67618b08f2aa954047b35e4f4d399cea2fa411065134935ded4fcec341b72d",
   "ipv4_before": 135,
   "ipv4_after": 136,
   "ipv6_before": 135,
   "ipv6_after": 136,
   "added_count": 2,
   "removed_count": 0,
   "ipv4_added": [
    "72.14.199.0/27"
   ],
   "ipv4_removed": [],
   "ipv6_added": [
    "2001:4860:4801:2043::/64"
   ],
   "ipv6_removed": [],
   "lists_truncated": false,
   "title": "Google special-purpose crawlers: 2 prefixes added, 0 removed",
   "detail": "The operator's own published list changed and this mirror followed it. Counts are exact; the lists are complete unless lists_truncated is true, in which case re-read /ip-ranges/google-special.json.",
   "affects": [
    "/ip-ranges/google-special.json",
    "/ip-ranges/all.json",
    "/ip-ranges/all.txt",
    "/status.json"
   ]
  },
  {
   "seq": 5,
   "at": "2026-09-01T20:40:09+00:00",
   "kind": "ip_ranges_changed",
   "source": "google-user-triggered",
   "operator": "google",
   "label": "Google user-triggered fetchers",
   "upstream": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers.json",
   "covers": [],
   "operator_published_at": "2026-09-01T14:45:48.000000",
   "sha256": "477db63c6cb203f9a93805b19d35cb16167c75befe93a1e806225df88ec7f9e7",
   "previous_sha256": "07361caaf20fc87fd8d3c082f8c779d96f7f152c9c5edcfd73b5fb6656c95ce9",
   "ipv4_before": 528,
   "ipv4_after": 529,
   "ipv6_before": 528,
   "ipv6_after": 529,
   "added_count": 2,
   "removed_count": 0,
   "ipv4_added": [
    "35.187.143.224/27"
   ],
   "ipv4_removed": [],
   "ipv6_added": [
    "2600:1900:0:63::/64"
   ],
   "ipv6_removed": [],
   "lists_truncated": false,
   "title": "Google user-triggered fetchers: 2 prefixes added, 0 removed",
   "detail": "The operator's own published list changed and this mirror followed it. Counts are exact; the lists are complete unless lists_truncated is true, in which case re-read /ip-ranges/google-user-triggered.json.",
   "affects": [
    "/ip-ranges/google-user-triggered.json",
    "/ip-ranges/all.json",
    "/ip-ranges/all.txt",
    "/status.json"
   ]
  },
  {
   "seq": 6,
   "at": "2026-09-01T20:40:10+00:00",
   "kind": "ip_ranges_changed",
   "source": "google-user-triggered-google",
   "operator": "google",
   "label": "Google user-triggered fetchers (Google-owned ranges)",
   "upstream": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers-google.json",
   "covers": [],
   "operator_published_at": "2026-09-01T14:45:54.000000",
   "sha256": "11ebb4ac8968d77f99af9d1e523ad0be7d7a87217e0f5e764e4b14f6a143dfeb",
   "previous_sha256": "fe0626007675ee18ccd5bc53e29601583bf0a4c3e1a1921146e4ff84c00af67d",
   "ipv4_before": 247,
   "ipv4_after": 248,
   "ipv6_before": 247,
   "ipv6_after": 248,
   "added_count": 2,
   "removed_count": 0,
   "ipv4_added": [
    "66.249.84.64/27"
   ],
   "ipv4_removed": [],
   "ipv6_added": [
    "2001:4860:4801:4063::/64"
   ],
   "ipv6_removed": [],
   "lists_truncated": false,
   "title": "Google user-triggered fetchers (Google-owned ranges): 2 prefixes added, 0 removed",
   "detail": "The operator's own published list changed and this mirror followed it. Counts are exact; the lists are complete unless lists_truncated is true, in which case re-read /ip-ranges/google-user-triggered-google.json.",
   "affects": [
    "/ip-ranges/google-user-triggered-google.json",
    "/ip-ranges/all.json",
    "/ip-ranges/all.txt",
    "/status.json"
   ]
  },
  {
   "seq": 7,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "adsbot-google",
   "name": "AdsBot-Google",
   "operator": "google",
   "category": "tool",
   "robots_token": "AdsBot-Google",
   "title": "New crawler record: AdsBot-Google",
   "detail": "Checks the quality of desktop landing pages for Google Ads. Google documents that it ignores the robots.txt * group with the ad publisher's permission, and obeys a group named for its own token.",
   "affects": [
    "/crawler/adsbot-google.json",
    "/crawler/adsbot-google.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 8,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "adsbot-google-mobile",
   "name": "AdsBot-Google-Mobile",
   "operator": "google",
   "category": "tool",
   "robots_token": "AdsBot-Google-Mobile",
   "title": "New crawler record: AdsBot-Google-Mobile",
   "detail": "The mobile-web landing page checker for Google Ads. Same rules as AdsBot-Google: the * group does not apply to it, its own token does.",
   "affects": [
    "/crawler/adsbot-google-mobile.json",
    "/crawler/adsbot-google-mobile.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 9,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "adsbot-google-mobile-apps",
   "name": "AdsBot-Google-Mobile-Apps",
   "operator": "google",
   "category": "tool",
   "robots_token": "AdsBot-Google-Mobile-Apps",
   "title": "New crawler record: AdsBot-Google-Mobile-Apps",
   "detail": "Checks Android app landing pages for Google Ads. It obeys a group named for its own token and, per Google, follows the AdsBot-Google rules otherwise.",
   "affects": [
    "/crawler/adsbot-google-mobile-apps.json",
    "/crawler/adsbot-google-mobile-apps.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 10,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_changed",
   "crawler": "ahrefsbot",
   "name": "AhrefsBot",
   "fields": [
    "ranges",
    "verify"
   ],
   "changed": {
    "ranges": {
     "before": "None",
     "after": "ahrefs-crawler"
    },
    "verify": {
     "before": "reverse-dns",
     "after": "published-ranges"
    }
   },
   "title": "AhrefsBot: ranges, verify changed",
   "detail": "A curated field moved. Every record carries the operator's own documentation URL, so a claim is one click from its source.",
   "affects": [
    "/crawler/ahrefsbot.json",
    "/crawler/ahrefsbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 11,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "ahrefssiteaudit",
   "name": "AhrefsSiteAudit",
   "operator": "ahrefs",
   "category": "seo",
   "robots_token": "AhrefsSiteAudit",
   "title": "New crawler record: AhrefsSiteAudit",
   "detail": "Ahrefs' site-audit crawler, separate from AhrefsBot. Ahrefs documents that it obeys robots.txt by default, and that a verified site owner can ask for it to be allowed to ignore robots.txt on their own site so the audit can see disallowed sections.",
   "affects": [
    "/crawler/ahrefssiteaudit.json",
    "/crawler/ahrefssiteaudit.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 12,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "aihitbot",
   "name": "aiHitBot",
   "operator": "aihit",
   "category": "dataset",
   "robots_token": "aiHitBot",
   "title": "New crawler record: aiHitBot",
   "detail": "aiHit's automated collector, building a company dataset from public company websites.",
   "affects": [
    "/crawler/aihitbot.json",
    "/crawler/aihitbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 13,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "aiwebindex",
   "name": "AIWebIndex",
   "operator": "lyrenth",
   "category": "ai-search",
   "robots_token": "AIWebIndex",
   "title": "New crawler record: AIWebIndex",
   "detail": "Builds an index of public pages and serves them to AI agents as extracted readable text, with attribution and a link back. Lyrenth publishes a crawler policy stating it does not train foundation models on what it collects and that it obeys robots.txt.",
   "affects": [
    "/crawler/aiwebindex.json",
    "/crawler/aiwebindex.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 14,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "andibot",
   "name": "Andibot",
   "operator": "andi",
   "category": "ai-search",
   "robots_token": "Andibot",
   "title": "New crawler record: Andibot",
   "detail": "The crawler for Andi, a small generative search assistant that summarises pages rather than listing them.",
   "affects": [
    "/crawler/andibot.json",
    "/crawler/andibot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 15,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "anomura",
   "name": "Anomura",
   "operator": "direqt",
   "category": "ai-search",
   "robots_token": "Anomura",
   "title": "New crawler record: Anomura",
   "detail": "Direqt's search crawler. It indexes the sites of Direqt's own publisher customers so their on-site chatbots can answer from them.",
   "affects": [
    "/crawler/anomura.json",
    "/crawler/anomura.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 16,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "apis-google",
   "name": "APIs-Google",
   "operator": "google",
   "category": "tool",
   "robots_token": "APIs-Google",
   "title": "New crawler record: APIs-Google",
   "detail": "Delivers push notifications for Google APIs to a webhook you registered. It is a special-case crawler: it ignores the robots.txt * group, because the fetch is a delivery to an address you asked it to deliver to.",
   "affects": [
    "/crawler/apis-google.json",
    "/crawler/apis-google.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 17,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "atlassian-bot",
   "name": "atlassian-bot",
   "operator": "atlassian",
   "category": "ai-search",
   "robots_token": "atlassian-bot",
   "title": "New crawler record: atlassian-bot",
   "detail": "Indexes a website so it can be searched and cited by Rovo, Atlassian's generative assistant inside Jira and Confluence. Atlassian's documentation walks a customer through editing robots.txt for it, which is as close to a compliance statement as this list gets.",
   "affects": [
    "/crawler/atlassian-bot.json",
    "/crawler/atlassian-bot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 18,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "awariorssbot",
   "name": "AwarioRssBot",
   "operator": "awario",
   "category": "dataset",
   "robots_token": "AwarioRssBot",
   "title": "New crawler record: AwarioRssBot",
   "detail": "The feed-reading half of Awario's pair, documented on the same page and under the same crawl-rate policy.",
   "affects": [
    "/crawler/awariorssbot.json",
    "/crawler/awariorssbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 19,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "awariosmartbot",
   "name": "AwarioSmartBot",
   "operator": "awario",
   "category": "dataset",
   "robots_token": "AwarioSmartBot",
   "title": "New crawler record: AwarioSmartBot",
   "detail": "Awario's brand-monitoring crawler. It documents one request per three seconds, honours Crawl-delay, and states it does not use consecutive IP blocks so identification is by user-agent only.",
   "affects": [
    "/crawler/awariosmartbot.json",
    "/crawler/awariosmartbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 20,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "barkrowler",
   "name": "Barkrowler",
   "operator": "babbar",
   "category": "seo",
   "robots_token": "barkrowler",
   "title": "New crawler record: Barkrowler",
   "detail": "Babbar's crawler, which builds the link graph behind their French-market SEO tooling.",
   "affects": [
    "/crawler/barkrowler.json",
    "/crawler/barkrowler.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 21,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "bedrockbot",
   "name": "bedrockbot",
   "operator": "amazon",
   "category": "ai-search",
   "robots_token": "bedrockbot",
   "title": "New crawler record: bedrockbot",
   "detail": "The web crawler an AWS customer points at URLs they chose, to build a knowledge base for a Bedrock application. AWS documents that it respects robots.txt and that the user-agent carries a per-customer suffix, so you can allow or refuse one customer's crawl by naming bedrockbot-UUID.",
   "affects": [
    "/crawler/bedrockbot.json",
    "/crawler/bedrockbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 22,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_changed",
   "crawler": "ccbot",
   "name": "CCBot",
   "fields": [
    "docs",
    "ranges",
    "verify"
   ],
   "changed": {
    "docs": {
     "before": "https://commoncrawl.org/faq",
     "after": "https://commoncrawl.org/ccbot"
    },
    "ranges": {
     "before": "None",
     "after": "commoncrawl-ccbot"
    },
    "verify": {
     "before": "none",
     "after": "published-ranges"
    }
   },
   "title": "CCBot: docs, ranges, verify changed",
   "detail": "A curated field moved. Every record carries the operator's own documentation URL, so a claim is one click from its source.",
   "affects": [
    "/crawler/ccbot.json",
    "/crawler/ccbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 23,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "chatgpt-agent",
   "name": "ChatGPT Agent",
   "operator": "openai",
   "category": "user-fetch",
   "robots_token": "ChatGPT-User",
   "title": "New crawler record: ChatGPT Agent",
   "detail": "ChatGPT's agent mode driving a real browser: it navigates and interacts with sites to finish a multi-step task a user gave it. OpenAI governs it with the ChatGPT-User token and the ChatGPT-User prefix list rather than a token of its own, so the robots rule and the address check are the same ones.",
   "affects": [
    "/crawler/chatgpt-agent.json",
    "/crawler/chatgpt-agent.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 24,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "cloudflare-autorag",
   "name": "Cloudflare-AutoRAG",
   "operator": "cloudflare",
   "category": "ai-search",
   "robots_token": "Cloudflare-AutoRAG",
   "title": "New crawler record: Cloudflare-AutoRAG",
   "detail": "The crawler behind Cloudflare's AI Search / AutoRAG, which indexes a website into a retrieval index for an application. Cloudflare's own documentation warns that a bot-blocking rule on your zone will also stop this crawler and tells you to allow-list it.",
   "affects": [
    "/crawler/cloudflare-autorag.json",
    "/crawler/cloudflare-autorag.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 25,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "cotoyogi",
   "name": "Cotoyogi",
   "operator": "rois",
   "category": "ai-training",
   "robots_token": "Cotoyogi",
   "title": "New crawler record: Cotoyogi",
   "detail": "A crawler run by ROIS-DS, a Japanese inter-university research organisation, collecting Japanese-language text for AI training. It publishes a crawler page in English and Japanese.",
   "affects": [
    "/crawler/cotoyogi.json",
    "/crawler/cotoyogi.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 26,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "crawl4ai",
   "name": "Crawl4AI",
   "operator": "crawl4ai",
   "category": "tool",
   "robots_token": "Crawl4AI",
   "title": "New crawler record: Crawl4AI",
   "detail": "An open-source LLM-oriented crawler and scraper library, run by whoever installs it. Like Scrapy, the default user-agent identifies the software and says nothing about who is behind the request.",
   "affects": [
    "/crawler/crawl4ai.json",
    "/crawler/crawl4ai.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 27,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "crawlspace",
   "name": "Crawlspace",
   "operator": "crawlspace",
   "category": "tool",
   "robots_token": "Crawlspace",
   "title": "New crawler record: Crawlspace",
   "detail": "A crawling platform: customers run their own crawls on it to feed agents, RAG pipelines and structured-data workflows. Like Firecrawl, the party behind any given request is the customer, not the platform.",
   "affects": [
    "/crawler/crawlspace.json",
    "/crawler/crawlspace.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 28,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "dataforseobot",
   "name": "DataForSeoBot",
   "operator": "dataforseo",
   "category": "seo",
   "robots_token": "DataForSeoBot",
   "title": "New crawler record: DataForSeoBot",
   "detail": "Builds the backlink and SERP datasets DataForSEO resells through its API, so one crawl reaches many downstream tools.",
   "affects": [
    "/crawler/dataforseobot.json",
    "/crawler/dataforseobot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 29,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "dotbot",
   "name": "DotBot",
   "operator": "moz",
   "category": "seo",
   "robots_token": "dotbot",
   "title": "New crawler record: DotBot",
   "detail": "Moz's crawler for Link Explorer. Moz documents that it respects robots.txt and that dotbot is the token to name.",
   "affects": [
    "/crawler/dotbot.json",
    "/crawler/dotbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 30,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "echoboxbot",
   "name": "EchoboxBot",
   "operator": "echobox",
   "category": "dataset",
   "robots_token": "EchoboxBot",
   "title": "New crawler record: EchoboxBot",
   "detail": "Collects data supporting Echobox's AI-driven social and email distribution products, which publishers use to schedule and target their own content.",
   "affects": [
    "/crawler/echoboxbot.json",
    "/crawler/echoboxbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 31,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "exasearchbot",
   "name": "ExaSearchBot",
   "operator": "exa",
   "category": "ai-search",
   "robots_token": "ExaSearchBot",
   "title": "New crawler record: ExaSearchBot",
   "detail": "Exa's crawler. It discovers and indexes public pages so they can be retrieved and cited through Exa's search API, which is one of the common retrieval backends behind agent frameworks.",
   "affects": [
    "/crawler/exasearchbot.json",
    "/crawler/exasearchbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 32,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "factset-spyderbot",
   "name": "Factset_spyderbot",
   "operator": "factset",
   "category": "ai-training",
   "robots_token": "Factset_spyderbot",
   "title": "New crawler record: Factset_spyderbot",
   "detail": "FactSet's crawler, collecting data used in AI model training for its financial data and analytics products.",
   "affects": [
    "/crawler/factset-spyderbot.json",
    "/crawler/factset-spyderbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 33,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "feedfetcher-google",
   "name": "FeedFetcher-Google",
   "operator": "google",
   "category": "tool",
   "robots_token": "FeedFetcher-Google",
   "title": "New crawler record: FeedFetcher-Google",
   "detail": "Crawls RSS and Atom feeds for Google News and WebSub. It is a user-triggered fetcher, and Google documents that those generally ignore robots.txt because a person asked for the fetch. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/feedfetcher-google.json",
    "/crawler/feedfetcher-google.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 34,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-agent",
   "name": "Google-Agent",
   "operator": "google",
   "category": "user-fetch",
   "robots_token": "Google-Agent",
   "title": "New crawler record: Google-Agent",
   "detail": "Agents hosted on Google infrastructure navigating the web and taking actions on a user's request. Google names one prefix list for it — user-triggered-agents.json — and is separately experimenting with Web Bot Auth under the identity https://agent.bot.goog.",
   "affects": [
    "/crawler/google-agent.json",
    "/crawler/google-agent.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 35,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-cws",
   "name": "Google-CWS",
   "operator": "google",
   "category": "tool",
   "robots_token": "Google-CWS",
   "title": "New crawler record: Google-CWS",
   "detail": "The Chrome Web Store fetcher. It requests the URLs a developer put in the metadata of a Chrome extension or theme. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/google-cws.json",
    "/crawler/google-cws.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 36,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-gemininotebook",
   "name": "Google-GeminiNotebook",
   "operator": "google",
   "category": "user-fetch",
   "robots_token": "Google-GeminiNotebook",
   "title": "New crawler record: Google-GeminiNotebook",
   "detail": "Fetches a URL a Gemini Notebook (formerly NotebookLM) user added as a source to their notebook. The former agent string Google-NotebookLM is documented as supported until August 2026. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/google-gemininotebook.json",
    "/crawler/google-gemininotebook.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 37,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-pinpoint",
   "name": "Google-Pinpoint",
   "operator": "google",
   "category": "user-fetch",
   "robots_token": "Google-Pinpoint",
   "title": "New crawler record: Google-Pinpoint",
   "detail": "Fetches individual URLs that a Pinpoint user — usually a journalist or researcher — added as a source to their own document collection. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/google-pinpoint.json",
    "/crawler/google-pinpoint.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 38,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-read-aloud",
   "name": "Google-Read-Aloud",
   "operator": "google",
   "category": "user-fetch",
   "robots_token": "Google-Read-Aloud",
   "title": "New crawler record: Google-Read-Aloud",
   "detail": "Fetches a page so Google can read it out loud with text-to-speech, at the moment a user asks. Formerly google-speakr. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/google-read-aloud.json",
    "/crawler/google-read-aloud.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 39,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-safety",
   "name": "Google-Safety",
   "operator": "google",
   "category": "tool",
   "robots_token": "Google-Safety",
   "title": "New crawler record: Google-Safety",
   "detail": "Google's abuse-investigation fetcher: malware review, phishing reports and similar. Google documents that it ignores robots.txt entirely, and a robots.txt rule for it does nothing.",
   "affects": [
    "/crawler/google-safety.json",
    "/crawler/google-safety.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 40,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "google-site-verification",
   "name": "Google-Site-Verification",
   "operator": "google",
   "category": "tool",
   "robots_token": "Google-Site-Verification",
   "title": "New crawler record: Google-Site-Verification",
   "detail": "Fetches the token file or meta tag that proves you own a site, when you click verify in Search Console. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/google-site-verification.json",
    "/crawler/google-site-verification.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 41,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "googlebot-video",
   "name": "Googlebot-Video",
   "operator": "google",
   "category": "search",
   "robots_token": "Googlebot-Video",
   "title": "New crawler record: Googlebot-Video",
   "detail": "The video half of Googlebot. It crawls video files and the pages around them for Google Video search, and it is matched by a robots.txt group for Googlebot as well as by its own token.",
   "affects": [
    "/crawler/googlebot-video.json",
    "/crawler/googlebot-video.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 42,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "googlemessages",
   "name": "GoogleMessages",
   "operator": "google",
   "category": "preview",
   "robots_token": "GoogleMessages",
   "title": "New crawler record: GoogleMessages",
   "detail": "Generates the link preview when somebody sends one of your URLs in Google Messages. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/googlemessages.json",
    "/crawler/googlemessages.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 43,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "googleother-image",
   "name": "GoogleOther-Image",
   "operator": "google",
   "category": "ai-training",
   "robots_token": "GoogleOther-Image",
   "title": "New crawler record: GoogleOther-Image",
   "detail": "The image variant of GoogleOther: one-off fetches by Google product and research teams that are not Search. It also answers to a GoogleOther group in robots.txt.",
   "affects": [
    "/crawler/googleother-image.json",
    "/crawler/googleother-image.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 44,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "googleother-video",
   "name": "GoogleOther-Video",
   "operator": "google",
   "category": "ai-training",
   "robots_token": "GoogleOther-Video",
   "title": "New crawler record: GoogleOther-Video",
   "detail": "The video variant of GoogleOther, used for internal Google fetches that do not belong to Search.",
   "affects": [
    "/crawler/googleother-video.json",
    "/crawler/googleother-video.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 45,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "googleproducer",
   "name": "GoogleProducer",
   "operator": "google",
   "category": "tool",
   "robots_token": "GoogleProducer",
   "title": "New crawler record: GoogleProducer",
   "detail": "Google Publisher Center: fetches the feeds a publisher explicitly supplied for Google News landing pages. Google publishes fetcher addresses in two files — user-triggered-fetchers.json and user-triggered-fetchers-google.json — and does not say per fetcher which one applies, so verification means checking both; this index mirrors both.",
   "affects": [
    "/crawler/googleproducer.json",
    "/crawler/googleproducer.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 46,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "icc-crawler",
   "name": "ICC-Crawler",
   "operator": "nict",
   "category": "ai-training",
   "robots_token": "ICC-Crawler",
   "title": "New crawler record: ICC-Crawler",
   "detail": "Operated by NICT, Japan's national information and communications research institute. The collected data supports AI research and, per the operator, is also provided to third parties including commercial companies.",
   "affects": [
    "/crawler/icc-crawler.json",
    "/crawler/icc-crawler.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 47,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "isscyberriskcrawler",
   "name": "ISSCyberRiskCrawler",
   "operator": "iss",
   "category": "ai-training",
   "robots_token": "ISSCyberRiskCrawler",
   "title": "New crawler record: ISSCyberRiskCrawler",
   "detail": "Crawls in order to train models that score a company's cyber risk. The ai.robots.txt dataset records the operator as not respecting robots.txt; ISS publishes no compliance statement of its own.",
   "affects": [
    "/crawler/isscyberriskcrawler.json",
    "/crawler/isscyberriskcrawler.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 48,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "kagibot",
   "name": "Kagibot",
   "operator": "kagi",
   "category": "search",
   "robots_token": "Kagibot",
   "title": "New crawler record: Kagibot",
   "detail": "The crawler for Kagi, a paid, ad-free search engine with its own index and its own assistant.",
   "affects": [
    "/crawler/kagibot.json",
    "/crawler/kagibot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 49,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "klaviyoaibot",
   "name": "KlaviyoAIBot",
   "operator": "klaviyo",
   "category": "ai-search",
   "robots_token": "KlaviyoAIBot",
   "title": "New crawler record: KlaviyoAIBot",
   "detail": "Fetches pages from domains a Klaviyo customer has explicitly connected to their own account, to power Klaviyo's Kai customer agent. It is scoped to connected domains rather than the open web.",
   "affects": [
    "/crawler/klaviyoaibot.json",
    "/crawler/klaviyoaibot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 50,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "laiondownloader",
   "name": "LAIONDownloader",
   "operator": "laion",
   "category": "dataset",
   "robots_token": "LAIONDownloader",
   "title": "New crawler record: LAIONDownloader",
   "detail": "LAION's downloader, used to materialise the image and text datasets the non-profit publishes for machine-learning research. LAION's own FAQ is the source for its robots.txt position.",
   "affects": [
    "/crawler/laiondownloader.json",
    "/crawler/laiondownloader.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 51,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "lightpanda",
   "name": "Lightpanda",
   "operator": "lightpanda",
   "category": "tool",
   "robots_token": "Lightpanda",
   "title": "New crawler record: Lightpanda",
   "detail": "A purpose-built headless browser for AI and automation — a runtime, not an operator. Whether robots.txt is honoured is left to whoever runs it, which is what its maintainers say themselves.",
   "affects": [
    "/crawler/lightpanda.json",
    "/crawler/lightpanda.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 52,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "linguee-bot",
   "name": "Linguee Bot",
   "operator": "linguee",
   "category": "ai-training",
   "robots_token": "Linguee Bot",
   "title": "New crawler record: Linguee Bot",
   "detail": "Gathers bilingual text for Linguee's translation corpus and the machine translation trained on it. Recorded in the ai.robots.txt dataset as not respecting robots.txt.",
   "affects": [
    "/crawler/linguee-bot.json",
    "/crawler/linguee-bot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 53,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "mediapartners-google",
   "name": "Mediapartners-Google",
   "operator": "google",
   "category": "tool",
   "robots_token": "Mediapartners-Google",
   "title": "New crawler record: Mediapartners-Google",
   "detail": "The AdSense crawler. It reads a page so AdSense can choose relevant ads for it, and it is a special-case crawler that ignores the robots.txt * group.",
   "affects": [
    "/crawler/mediapartners-google.json",
    "/crawler/mediapartners-google.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 54,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "meta-webindexer",
   "name": "Meta-WebIndexer",
   "operator": "meta",
   "category": "ai-search",
   "robots_token": "Meta-WebIndexer",
   "title": "New crawler record: Meta-WebIndexer",
   "detail": "Per Meta's crawler documentation, Meta-WebIndexer navigates the web to improve the quality of Meta AI's search results. It is a third Meta token alongside Meta-ExternalAgent and Meta-ExternalFetcher, and the newest of them.",
   "affects": [
    "/crawler/meta-webindexer.json",
    "/crawler/meta-webindexer.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 55,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "mj12bot",
   "name": "MJ12bot",
   "operator": "majestic",
   "category": "seo",
   "robots_token": "MJ12bot",
   "title": "New crawler record: MJ12bot",
   "detail": "Majestic's link-graph crawler, run as a distributed community project. Majestic states plainly that it cannot restrict the bot to a fixed set of addresses, and offers a pre-arranged ident string in the request headers instead.",
   "affects": [
    "/crawler/mj12bot.json",
    "/crawler/mj12bot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 56,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "mojeekbot",
   "name": "MojeekBot",
   "operator": "mojeek",
   "category": "search",
   "robots_token": "MojeekBot",
   "title": "New crawler record: MojeekBot",
   "detail": "Mojeek's crawler. Mojeek runs one of the few genuinely independent web indexes — not a front end over Bing or Google — so it is one of the few blocks that removes you from an index nobody else can put you back into. Its documentation states it obeys the first record whose User-Agent contains MojeekBot, falling back to *.",
   "affects": [
    "/crawler/mojeekbot.json",
    "/crawler/mojeekbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 57,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "panscient",
   "name": "Panscient",
   "operator": "panscient",
   "category": "dataset",
   "robots_token": "panscient.com",
   "title": "New crawler record: Panscient",
   "detail": "Compiles structured data about businesses and business professionals using machine learning. Panscient's FAQ states it obeys robots.txt.",
   "affects": [
    "/crawler/panscient.json",
    "/crawler/panscient.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 58,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "phindbot",
   "name": "PhindBot",
   "operator": "phind",
   "category": "ai-search",
   "robots_token": "PhindBot",
   "title": "New crawler record: PhindBot",
   "detail": "Phind is an answer engine for developers that combines live web search with its own models. This is the crawler behind those answers.",
   "affects": [
    "/crawler/phindbot.json",
    "/crawler/phindbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 59,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "pinterestbot",
   "name": "Pinterestbot",
   "operator": "pinterest",
   "category": "search",
   "robots_token": "Pinterestbot",
   "title": "New crawler record: Pinterestbot",
   "detail": "Pinterest's crawler. It indexes pages so people can find them on Pinterest and re-reads product pages to keep price and title on a Pin current. Pinterest states that content it crawls is not used to train their Canvas image generation model.",
   "affects": [
    "/crawler/pinterestbot.json",
    "/crawler/pinterestbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 60,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "poseidon-research-crawler",
   "name": "Poseidon Research Crawler",
   "operator": "poseidon",
   "category": "ai-training",
   "robots_token": "Poseidon Research Crawler",
   "title": "New crawler record: Poseidon Research Crawler",
   "detail": "A crawler run by Poseidon Research, a lab working on interpretability research for AI systems.",
   "affects": [
    "/crawler/poseidon-research-crawler.json",
    "/crawler/poseidon-research-crawler.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 61,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "qualifiedbot",
   "name": "QualifiedBot",
   "operator": "qualified",
   "category": "ai-search",
   "robots_token": "QualifiedBot",
   "title": "New crawler record: QualifiedBot",
   "detail": "Analyses a customer's website so Qualified's AI sales chatbots can answer questions about it in context.",
   "affects": [
    "/crawler/qualifiedbot.json",
    "/crawler/qualifiedbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 62,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "quillbot",
   "name": "QuillBot",
   "operator": "quillbot",
   "category": "ai-training",
   "robots_token": "QuillBot",
   "title": "New crawler record: QuillBot",
   "detail": "Operated by QuillBot as part of its writing, paraphrasing and AI-detection products. The dataset also records a second token, quillbot.com, for the same operator.",
   "affects": [
    "/crawler/quillbot.json",
    "/crawler/quillbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 63,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "qwantbot",
   "name": "Qwantbot",
   "operator": "qwant",
   "category": "search",
   "robots_token": "Qwantbot",
   "title": "New crawler record: Qwantbot",
   "detail": "Qwant's crawler. Qwant documents that the string Qwantbot always appears in its user-agents whatever the crawler version, which is what makes a substring match safe here.",
   "affects": [
    "/crawler/qwantbot.json",
    "/crawler/qwantbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 64,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "qwantbot-news",
   "name": "Qwantbot-news",
   "operator": "qwant",
   "category": "search",
   "robots_token": "Qwantbot-news",
   "title": "New crawler record: Qwantbot-news",
   "detail": "The news variant of Qwant's crawler, documented alongside the main one and carrying the same Qwantbot substring.",
   "affects": [
    "/crawler/qwantbot-news.json",
    "/crawler/qwantbot-news.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 65,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "reflectionbot",
   "name": "Reflectionbot",
   "operator": "reflection",
   "category": "ai-training",
   "robots_token": "Reflectionbot",
   "title": "New crawler record: Reflectionbot",
   "detail": "An undocumented crawler whose user-agent links to Reflection AI, a company building AI models. The link in the user-agent is the only public statement of purpose that exists.",
   "affects": [
    "/crawler/reflectionbot.json",
    "/crawler/reflectionbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 66,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "rogerbot",
   "name": "rogerbot",
   "operator": "moz",
   "category": "seo",
   "robots_token": "rogerbot",
   "title": "New crawler record: rogerbot",
   "detail": "Moz's Campaign crawler, which audits a site its own owner registered. Moz states there is no IP range for it — identification is by user-agent only.",
   "affects": [
    "/crawler/rogerbot.json",
    "/crawler/rogerbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 67,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "sbintuitionsbot",
   "name": "SBIntuitionsBot",
   "operator": "sbintuitions",
   "category": "ai-training",
   "robots_token": "SBIntuitionsBot",
   "title": "New crawler record: SBIntuitionsBot",
   "detail": "SB Intuitions is SoftBank's Japanese LLM lab; this crawler gathers data used in that model development and in information analysis. The operator publishes a dedicated bot page.",
   "affects": [
    "/crawler/sbintuitionsbot.json",
    "/crawler/sbintuitionsbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 68,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "screaming-frog-seo-spider",
   "name": "Screaming Frog SEO Spider",
   "operator": "screamingfrog",
   "category": "tool",
   "robots_token": "Screaming Frog SEO Spider",
   "title": "New crawler record: Screaming Frog SEO Spider",
   "detail": "Not an operator: desktop crawling software that anybody can point at any site. The default user-agent identifies the tool, not who is running it, and the operator of the moment is whoever pressed start.",
   "affects": [
    "/crawler/screaming-frog-seo-spider.json",
    "/crawler/screaming-frog-seo-spider.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 69,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "semrushbot-ba",
   "name": "SemrushBot-BA",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SemrushBot-BA",
   "title": "New crawler record: SemrushBot-BA",
   "detail": "The Backlink Audit crawler. It re-checks links pointing at a customer's site, which means it lands on the sites doing the linking.",
   "affects": [
    "/crawler/semrushbot-ba.json",
    "/crawler/semrushbot-ba.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 70,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "semrushbot-esi",
   "name": "SemrushBot-ESI",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SemrushBot-ESI",
   "title": "New crawler record: SemrushBot-ESI",
   "detail": "The crawler for Semrush Enterprise Site Intelligence, the enterprise tier's own site analysis.",
   "affects": [
    "/crawler/semrushbot-esi.json",
    "/crawler/semrushbot-esi.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 71,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "semrushbot-ft",
   "name": "SemrushBot-FT",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SemrushBot-FT",
   "title": "New crawler record: SemrushBot-FT",
   "detail": "Fetches full text for the Plagiarism Checker and similar text-comparison tools.",
   "affects": [
    "/crawler/semrushbot-ft.json",
    "/crawler/semrushbot-ft.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 72,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "semrushbot-si",
   "name": "SemrushBot-SI",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SemrushBot-SI",
   "title": "New crawler record: SemrushBot-SI",
   "detail": "Fetches pages for the On Page SEO Checker and similar advisory tools.",
   "affects": [
    "/crawler/semrushbot-si.json",
    "/crawler/semrushbot-si.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 73,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "semrushbot-swa",
   "name": "SemrushBot-SWA",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SemrushBot-SWA",
   "title": "New crawler record: SemrushBot-SWA",
   "detail": "Checks whether a URL is reachable, for the SEO Writing Assistant. One request per URL a writer references, not a crawl.",
   "affects": [
    "/crawler/semrushbot-swa.json",
    "/crawler/semrushbot-swa.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 74,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "seokicks",
   "name": "SEOkicks",
   "operator": "seokicks",
   "category": "seo",
   "robots_token": "SEOkicks",
   "title": "New crawler record: SEOkicks",
   "detail": "A German backlink index. Its documentation names SEOkicks as the user-agent to use in robots.txt.",
   "affects": [
    "/crawler/seokicks.json",
    "/crawler/seokicks.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 75,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "serpstatbot",
   "name": "serpstatbot",
   "operator": "serpstat",
   "category": "seo",
   "robots_token": "serpstatbot",
   "title": "New crawler record: serpstatbot",
   "detail": "Serpstat's backlink crawler. It documents support for Crawl-delay up to 20 seconds, including a delay set on the * group.",
   "affects": [
    "/crawler/serpstatbot.json",
    "/crawler/serpstatbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 76,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "shapbot",
   "name": "ShapBot",
   "operator": "parallel",
   "category": "ai-search",
   "robots_token": "ShapBot",
   "title": "New crawler record: ShapBot",
   "detail": "Parallel's crawler. It collects and structures web content to power the search, extraction and deep-research APIs that Parallel sells to agent builders.",
   "affects": [
    "/crawler/shapbot.json",
    "/crawler/shapbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 77,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "sidetrade-indexer-bot",
   "name": "Sidetrade indexer bot",
   "operator": "sidetrade",
   "category": "ai-training",
   "robots_token": "Sidetrade indexer bot",
   "title": "New crawler record: Sidetrade indexer bot",
   "detail": "Sidetrade extracts web data for a range of uses including training its AI products for order-to-cash and customer-data work.",
   "affects": [
    "/crawler/sidetrade-indexer-bot.json",
    "/crawler/sidetrade-indexer-bot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 78,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "siteauditbot",
   "name": "SiteAuditBot",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SiteAuditBot",
   "title": "New crawler record: SiteAuditBot",
   "detail": "Semrush's Site Audit crawler: it walks a site a customer owns and reports technical SEO problems. Semrush names it as the token to block for that product.",
   "affects": [
    "/crawler/siteauditbot.json",
    "/crawler/siteauditbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 79,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "slackbot",
   "name": "Slackbot",
   "operator": "slack",
   "category": "preview",
   "robots_token": "Slackbot",
   "title": "New crawler record: Slackbot",
   "detail": "The other half of Slack's pair: the agent that reads robots.txt and handles Slack's non-unfurl fetches. Slack documents both strings on one page.",
   "affects": [
    "/crawler/slackbot.json",
    "/crawler/slackbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 80,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "slackbot-linkexpanding",
   "name": "Slackbot-LinkExpanding",
   "operator": "slack",
   "category": "preview",
   "robots_token": "Slackbot-LinkExpanding",
   "title": "New crawler record: Slackbot-LinkExpanding",
   "detail": "Fetches a page to build the unfurl card when somebody pastes your link into Slack. One paste, one fetch.",
   "affects": [
    "/crawler/slackbot-linkexpanding.json",
    "/crawler/slackbot-linkexpanding.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 81,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "splitsignalbot",
   "name": "SplitSignalBot",
   "operator": "semrush",
   "category": "seo",
   "robots_token": "SplitSignalBot",
   "title": "New crawler record: SplitSignalBot",
   "detail": "Runs SEO A/B tests on a customer's own site with the SplitSignal tool.",
   "affects": [
    "/crawler/splitsignalbot.json",
    "/crawler/splitsignalbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 82,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "terracotta",
   "name": "TerraCotta",
   "operator": "ceramic",
   "category": "ai-search",
   "robots_token": "TerraCotta",
   "title": "New crawler record: TerraCotta",
   "detail": "Ceramic AI's crawler, which indexes public content for a web-scale search API aimed at LLMs and agents.",
   "affects": [
    "/crawler/terracotta.json",
    "/crawler/terracotta.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 83,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "thinkbot",
   "name": "Thinkbot",
   "operator": "thinkbot",
   "category": "dataset",
   "robots_token": "Thinkbot",
   "title": "New crawler record: Thinkbot",
   "detail": "Collects pages for analysis of how sites are adopting AI and automation. The ai.robots.txt dataset records the operator as not respecting robots.txt.",
   "affects": [
    "/crawler/thinkbot.json",
    "/crawler/thinkbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 84,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "velenpublicwebcrawler",
   "name": "VelenPublicWebCrawler",
   "operator": "hunter",
   "category": "dataset",
   "robots_token": "VelenPublicWebCrawler",
   "title": "New crawler record: VelenPublicWebCrawler",
   "detail": "Hunter's crawler, written in Go, building business datasets and machine-learning models from public pages. Its page states it follows robots.txt and meta directives and never fetches more than one page every two seconds.",
   "affects": [
    "/crawler/velenpublicwebcrawler.json",
    "/crawler/velenpublicwebcrawler.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 85,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "wpbot",
   "name": "wpbot",
   "operator": "quantumcloud",
   "category": "tool",
   "robots_token": "wpbot",
   "title": "New crawler record: wpbot",
   "detail": "Supports the AI Chatbot for WordPress plugin: it reads pages so the plugin can answer from a site's own content. The operator provides an opt-out through a form rather than through robots.txt.",
   "affects": [
    "/crawler/wpbot.json",
    "/crawler/wpbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 86,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yak",
   "name": "YaK",
   "operator": "meltwater",
   "category": "dataset",
   "robots_token": "YaK",
   "title": "New crawler record: YaK",
   "detail": "Meltwater's crawler, feeding the live data stream behind its media-monitoring and consumer-intelligence suite.",
   "affects": [
    "/crawler/yak.json",
    "/crawler/yak.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 87,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexadditional",
   "name": "YandexAdditional",
   "operator": "yandex",
   "category": "ai-training",
   "robots_token": "YandexAdditional",
   "title": "New crawler record: YandexAdditional",
   "detail": "The token that controls whether already-indexed pages may appear in Search with Yandex AI answers. Yandex's table says it makes no indexing requests of its own — it exists so a site can opt out of the generative answer without leaving the index.",
   "affects": [
    "/crawler/yandexadditional.json",
    "/crawler/yandexadditional.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 88,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexadditionalbot",
   "name": "YandexAdditionalBot",
   "operator": "yandex",
   "category": "ai-training",
   "robots_token": "YandexAdditionalBot",
   "title": "New crawler record: YandexAdditionalBot",
   "detail": "The second token Yandex publishes for the same AI-answers opt-out. Both names appear in Yandex's own robot list, so a robots.txt that names only one of them is half a policy.",
   "affects": [
    "/crawler/yandexadditionalbot.json",
    "/crawler/yandexadditionalbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 89,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexblogs",
   "name": "YandexBlogs",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexBlogs",
   "title": "New crawler record: YandexBlogs",
   "detail": "Yandex's blog-search robot; it indexes post comments as well as posts.",
   "affects": [
    "/crawler/yandexblogs.json",
    "/crawler/yandexblogs.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 90,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexcalendar",
   "name": "YandexCalendar",
   "operator": "yandex",
   "category": "user-fetch",
   "robots_token": "YandexCalendar",
   "title": "New crawler record: YandexCalendar",
   "detail": "Downloads calendar files a user subscribed to. Yandex notes these files are often in directories that are disallowed for indexing, which is why the general rules are not applied.",
   "affects": [
    "/crawler/yandexcalendar.json",
    "/crawler/yandexcalendar.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 91,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexcombot",
   "name": "YandexComBot",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexComBot",
   "title": "New crawler record: YandexComBot",
   "detail": "Indexes content for Yandex search in languages other than Russian. Yandex documents that it can index content when there is no explicit robot-specific restriction — a * group is not one.",
   "affects": [
    "/crawler/yandexcombot.json",
    "/crawler/yandexcombot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 92,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexdirect",
   "name": "YandexDirect",
   "operator": "yandex",
   "category": "tool",
   "robots_token": "YandexDirect",
   "title": "New crawler record: YandexDirect",
   "detail": "Reads the content of Yandex Advertising Network partner pages to work out their topic so relevant ads can be matched. Documented as not taking the general robots.txt rules into account.",
   "affects": [
    "/crawler/yandexdirect.json",
    "/crawler/yandexdirect.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 93,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexfavicons",
   "name": "YandexFavicons",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexFavicons",
   "title": "New crawler record: YandexFavicons",
   "detail": "Downloads your favicon so Yandex can show it beside your result. Documented as not taking the general robots.txt rules into account.",
   "affects": [
    "/crawler/yandexfavicons.json",
    "/crawler/yandexfavicons.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 94,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandeximages",
   "name": "YandexImages",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexImages",
   "title": "New crawler record: YandexImages",
   "detail": "Indexes images for Yandex Images. Yandex's robot table marks it as taking the general robots.txt rules into account.",
   "affects": [
    "/crawler/yandeximages.json",
    "/crawler/yandeximages.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 95,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexmarket",
   "name": "YandexMarket",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexMarket",
   "title": "New crawler record: YandexMarket",
   "detail": "The robot behind Yandex Market, Yandex's shopping comparison service. Version 1.0 is documented as obeying the general rules; version 2.0 is documented as not.",
   "affects": [
    "/crawler/yandexmarket.json",
    "/crawler/yandexmarket.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 96,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexmedia",
   "name": "YandexMedia",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexMedia",
   "title": "New crawler record: YandexMedia",
   "detail": "Indexes multimedia data for Yandex. Takes the general robots.txt rules into account.",
   "affects": [
    "/crawler/yandexmedia.json",
    "/crawler/yandexmedia.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 97,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexmetrika",
   "name": "YandexMetrika",
   "operator": "yandex",
   "category": "tool",
   "robots_token": "YandexMetrika",
   "title": "New crawler record: YandexMetrika",
   "detail": "Yandex Metrica's own fetcher. Two of its versions — the 2.0 yabs01 availability checker and the 4.0 CSS cache for Webvisor — are documented in Yandex's table as not using robots.txt at all.",
   "affects": [
    "/crawler/yandexmetrika.json",
    "/crawler/yandexmetrika.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 98,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexmobilebot",
   "name": "YandexMobileBot",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexMobileBot",
   "title": "New crawler record: YandexMobileBot",
   "detail": "Decides whether a page's layout is suitable for mobile devices. Yandex's table marks it as NOT taking the general robots.txt rules into account, so a * group does not stop it — a group named YandexMobileBot does.",
   "affects": [
    "/crawler/yandexmobilebot.json",
    "/crawler/yandexmobilebot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 99,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexrenderresourcesbot",
   "name": "YandexRenderResourcesBot",
   "operator": "yandex",
   "category": "search",
   "robots_token": "YandexRenderResourcesBot",
   "title": "New crawler record: YandexRenderResourcesBot",
   "detail": "Loads the CSS, JavaScript and images Yandex needs to render a page. Yandex documents the exact rule: it ignores robots.txt for a resource when the HTML page using it is allowed, and does not fetch the resource when that page is disallowed.",
   "affects": [
    "/crawler/yandexrenderresourcesbot.json",
    "/crawler/yandexrenderresourcesbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  },
  {
   "seq": 100,
   "at": "2026-09-01T20:40:11+00:00",
   "kind": "crawler_added",
   "crawler": "yandexscreenshotbot",
   "name": "YandexScreenshotBot",
   "operator": "yandex",
   "category": "tool",
   "robots_token": "YandexScreenshotBot",
   "title": "New crawler record: YandexScreenshotBot",
   "detail": "Takes a screenshot of a page. Documented as not taking the general robots.txt rules into account.",
   "affects": [
    "/crawler/yandexscreenshotbot.json",
    "/crawler/yandexscreenshotbot.md",
    "/data/agents.json",
    "/data/agents.csv"
   ]
  }
 ],
 "sources_checked_at": "2026-09-01T21:38:57+00:00",
 "sources_ok": 15,
 "sources_total": 15,
 "stale_sources": [],
 "stale_note": "Every upstream answered on the last refresh.",
 "how_to_poll": {
  "upstreams_refetched_every": "6 hours",
  "minimum_useful_interval_seconds": 21600,
  "why": "The 15 operator endpoints are re-fetched every six hours and this feed cannot move faster than they do. Polling more often is not rewarded with more news; it returns this same cursor. Nothing here is rate limited — the request is simply not worth your budget.",
  "cheapest_check": "Send the ETag from your last response as If-None-Match. Unchanged is a 304 with no body.",
  "conditional_get": "ETag (strong) and Last-Modified on every answer; If-None-Match and If-Modified-Since both honoured, here and on every static document on this host."
 },
 "documents": {
  "/status.json": "per-source freshness, the same facts as stale_sources",
  "/ip-ranges/all.json": "the union of every mirrored prefix, grouped by source",
  "/data/agents.json": "every crawler record",
  "/openapi.json": "this endpoint described formally"
 },
 "license": "CC0-1.0"
}