{
 "generated_at": "2026-09-03T02:39:29+00:00",
 "note": "Case-insensitive substring alternations, already escaped. A user-agent match is a claim, not a proof: pair it with /ip-ranges/ or reverse DNS before you trust it.",
 "all": "(AdsBot\\-Google|AdsBot\\-Google\\-Mobile|AdsBot\\-Google\\-Mobile\\-Apps|AhrefsBot|AhrefsSiteAudit|AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AIWebIndex|Amazonbot|Andibot|Anomura|anthropic\\-ai|APIs\\-Google|Applebot|archive\\.org_bot|atlassian\\-bot|AwarioRssBot|AwarioSmartBot|Baiduspider|barkrowler|bedrockbot|bingbot|Bytespider|CCBot|ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-SearchBot|Claude\\-User|Claude\\-Web|ClaudeBot|Cloudflare\\-AutoRAG|cohere\\-ai|cohere\\-training\\-data\\-crawler|Cotoyogi|Crawl4AI|Crawlspace|DataForSeoBot|Diffbot|dotbot|DuckAssistBot|DuckDuckBot|EchoboxBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FeedFetcher\\-Google|FirecrawlAgent|Google\\-Agent|Google\\-CloudVertexBot|Google\\-CWS|Google\\-GeminiNotebook|Google\\-InspectionTool|Google\\-Pinpoint|Google\\-Read\\-Aloud|Google\\-Safety|Google\\-Site\\-Verification|Googlebot|Googlebot\\-Image|Googlebot\\-News|Googlebot\\-Video|GoogleMessages|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GoogleProducer|GPTBot|ia_archiver|ICC\\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kagibot|KlaviyoAIBot|LAIONDownloader|Lightpanda|Linguee\\ Bot|Mediapartners\\-Google|meta\\-externalagent|meta\\-externalfetcher|Meta\\-WebIndexer|MistralAI\\-User|MJ12bot|MojeekBot|OAI\\-SearchBot|omgili|omgilibot|panscient\\.com|Perplexity\\-User|PerplexityBot|PetalBot|PhindBot|Pinterestbot|Poseidon\\ Research\\ Crawler|QualifiedBot|QuillBot|Qwantbot|Qwantbot\\-news|Reflectionbot|rogerbot|SBIntuitionsBot|Scrapy|Screaming\\ Frog\\ SEO\\ Spider|SemrushBot|SemrushBot\\-BA|SemrushBot\\-ESI|SemrushBot\\-FT|SemrushBot\\-OCOB|SemrushBot\\-SI|SemrushBot\\-SWA|SEOkicks|serpstatbot|SeznamBot|ShapBot|Sidetrade\\ indexer\\ bot|SiteAuditBot|Slackbot|Slackbot\\-LinkExpanding|SplitSignalBot|Storebot\\-Google|TerraCotta|Thinkbot|TikTokSpider|Timpibot|VelenPublicWebCrawler|Webzio\\-Extended|wpbot|YaK|YandexAdditional|YandexAdditionalBot|YandexBlogs|YandexBot|YandexCalendar|YandexComBot|YandexDirect|YandexFavicons|YandexImages|YandexMarket|YandexMedia|YandexMetrika|YandexMobileBot|YandexRenderResourcesBot|YandexScreenshotBot|YandexVideo|YandexWebmaster|Yeti|YouBot)",
 "ai_only": "(AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AIWebIndex|Amazonbot|Andibot|Anomura|anthropic\\-ai|atlassian\\-bot|AwarioRssBot|AwarioSmartBot|bedrockbot|Bytespider|CCBot|ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-SearchBot|Claude\\-User|Claude\\-Web|ClaudeBot|Cloudflare\\-AutoRAG|cohere\\-ai|cohere\\-training\\-data\\-crawler|Cotoyogi|Diffbot|DuckAssistBot|EchoboxBot|ExaSearchBot|FacebookBot|Factset_spyderbot|Google\\-Agent|Google\\-CloudVertexBot|Google\\-GeminiNotebook|Google\\-Pinpoint|Google\\-Read\\-Aloud|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GPTBot|ICC\\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|KlaviyoAIBot|LAIONDownloader|Linguee\\ Bot|meta\\-externalagent|meta\\-externalfetcher|Meta\\-WebIndexer|MistralAI\\-User|OAI\\-SearchBot|omgili|omgilibot|panscient\\.com|Perplexity\\-User|PerplexityBot|PhindBot|Poseidon\\ Research\\ Crawler|QualifiedBot|QuillBot|Reflectionbot|SBIntuitionsBot|SemrushBot\\-OCOB|ShapBot|Sidetrade\\ indexer\\ bot|TerraCotta|Thinkbot|TikTokSpider|VelenPublicWebCrawler|Webzio\\-Extended|YaK|YandexAdditional|YandexAdditionalBot|YandexCalendar|YouBot)",
 "by_category": {
  "ai-training": "(anthropic\\-ai|Bytespider|ClaudeBot|cohere\\-training\\-data\\-crawler|Cotoyogi|FacebookBot|Factset_spyderbot|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GPTBot|ICC\\-Crawler|ISSCyberRiskCrawler|Linguee\\ Bot|meta\\-externalagent|Poseidon\\ Research\\ Crawler|QuillBot|Reflectionbot|SBIntuitionsBot|SemrushBot\\-OCOB|Sidetrade\\ indexer\\ bot|TikTokSpider|Webzio\\-Extended|YandexAdditional|YandexAdditionalBot)",
  "ai-search": "(AIWebIndex|Amazonbot|Andibot|Anomura|atlassian\\-bot|bedrockbot|Claude\\-SearchBot|Claude\\-Web|Cloudflare\\-AutoRAG|DuckAssistBot|ExaSearchBot|Google\\-CloudVertexBot|KlaviyoAIBot|Meta\\-WebIndexer|OAI\\-SearchBot|PerplexityBot|PhindBot|QualifiedBot|ShapBot|TerraCotta|YouBot)",
  "user-fetch": "(ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-User|cohere\\-ai|Google\\-Agent|Google\\-GeminiNotebook|Google\\-Pinpoint|Google\\-Read\\-Aloud|meta\\-externalfetcher|MistralAI\\-User|Perplexity\\-User|YandexCalendar)",
  "search": "(Applebot|Baiduspider|bingbot|DuckDuckBot|Googlebot|Googlebot\\-Image|Googlebot\\-News|Googlebot\\-Video|Kagibot|MojeekBot|PetalBot|Pinterestbot|Qwantbot|Qwantbot\\-news|SeznamBot|Storebot\\-Google|Timpibot|YandexBlogs|YandexBot|YandexComBot|YandexFavicons|YandexImages|YandexMarket|YandexMedia|YandexMobileBot|YandexRenderResourcesBot|YandexVideo|Yeti)",
  "tool": "(AdsBot\\-Google|AdsBot\\-Google\\-Mobile|AdsBot\\-Google\\-Mobile\\-Apps|APIs\\-Google|Crawl4AI|Crawlspace|FeedFetcher\\-Google|FirecrawlAgent|Google\\-CWS|Google\\-InspectionTool|Google\\-Safety|Google\\-Site\\-Verification|GoogleProducer|Lightpanda|Mediapartners\\-Google|Scrapy|Screaming\\ Frog\\ SEO\\ Spider|wpbot|YandexDirect|YandexMetrika|YandexScreenshotBot|YandexWebmaster)",
  "dataset": "(AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AwarioRssBot|AwarioSmartBot|CCBot|Diffbot|EchoboxBot|ImagesiftBot|img2dataset|LAIONDownloader|omgili|omgilibot|panscient\\.com|Thinkbot|VelenPublicWebCrawler|YaK)",
  "preview": "(facebookexternalhit|GoogleMessages|Slackbot|Slackbot\\-LinkExpanding)",
  "seo": "(AhrefsBot|AhrefsSiteAudit|barkrowler|DataForSeoBot|dotbot|MJ12bot|rogerbot|SemrushBot|SemrushBot\\-BA|SemrushBot\\-ESI|SemrushBot\\-FT|SemrushBot\\-SI|SemrushBot\\-SWA|SEOkicks|serpstatbot|SiteAuditBot|SplitSignalBot)",
  "archive": "(archive\\.org_bot|ia_archiver)"
 },
 "links": [
  {
   "rel": "self",
   "href": "https://www.pathwren.workers.dev/data/ua-regex.json",
   "type": "application/json"
  },
  {
   "rel": "changes",
   "href": "https://www.pathwren.workers.dev/changes.json?since=110",
   "type": "application/json",
   "title": "What changed since your cursor — poll this instead of re-downloading this document",
   "cursor_param": "since",
   "head_cursor": 110,
   "min_poll_seconds": 21600,
   "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag."
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/data/agents.json",
   "type": "application/json",
   "title": "Every crawler record in one file"
  },
  {
   "rel": "related",
   "href": "https://www.pathwren.workers.dev/ip-ranges/all.json",
   "type": "application/json",
   "title": "Every operator-published prefix, unioned and grouped by source"
  },
  {
   "rel": "service-desc",
   "href": "https://www.pathwren.workers.dev/openapi.json",
   "type": "application/json",
   "title": "Every read endpoint, described formally"
  }
 ],
 "observed_at_edge": "(402explorer|a2a\\-directory\\-discovery|a2a\\-directory\\-liveness|A2A\\-Indexer|A2A\\-Registry|A2A\\-Registry\\-HealthCheck|A2A\\-Registry\\-Scanner|A2A\\-Registry\\-Smoke|a2a\\-security\\-research\\-crawler|AetherLink\\-Public\\-Agent\\-Card\\-Policy\\-Check|AetherLink\\-Public\\-Discovery\\-Evidence|AetherLinkDiscoveryEvidence|AffsignalCrawler|AgenstryBot|agent\\-guild\\-scout|agent\\-ready\\-scanner|agent\\-tools\\.cloud\\-crawler|agent\\-world\\-probe|AgentAlmanac\\-Snapshot|AgentCatalogBot|AgentDisco|AgentGrade|AgentIndexBot|AgentPointsDirectoryEnricher|AgentReputationBot|Agentry\\-Registry|AgentSure\\-MCPScan|AgentTrust\\-Monitor|AhrefsBot|ai\\-crawler\\-logs|ai\\-crawler\\-robots|ai\\-crawler\\-verify|aisec\\-registry|AIVE\\-MCP\\-Discover|AIVE\\-MCP\\-EndpointProbe|Amazonbot|api\\-forge\\-mcp\\-index|APIEvangelist|apievangelist\\-domain\\-probe|apievangelist\\-security\\-probe|apis\\.io\\-submit|archive\\.org_bot|ardcrawl|AzureAI\\-SearchBot|bingbot|Bytespider|CensusBot|Chrome|ClaudeBot|Conway\\-Replicatio\\-r91\\-strict\\-a2a\\-probe|Discordbot|DuckDuckBot|EndpointAudit|Enerlio|exaforce\\-mcprep|FreePublicAPIs|frndOS|GF\\-Agent\\-Toll\\-Outbound|GF\\-Agent\\-Toll\\-Remediation|GolemreachTrustBot|Googlebot|GoogleOther|GPTBot|GraphAdvocate\\-Outreach|gtm\\-engine|guild\\-reachability\\-probe|hhvm\\-internal|hultra\\-link|invinoveritas\\-handshake|io\\.verifymcp|itinai\\-importer|jscrawler|lastseen\\-schema\\-probe|llm4agents\\-cimd\\-audit|loop\\-mcp\\-catalog\\-fetch|MCP\\-Catalog|mcp\\-checker|mcp\\-drift\\-monitor|mcp\\-observatory|mcp\\-registry|mcp\\-rugpull\\-research|mcp\\-schema\\-archive|MCP\\-Stats\\-Prober|mcp\\-uptime|mcp2\\-research|mcpbeat|MCPCatalogSync|mcpcheck|mcpgrade\\-probe|mcpi|MCPMeter|mcpqueen\\-grader|mcpscan|MCPWatch|MCPWitness|measure\\-mcp\\-schema|meta\\-externalagent|movanas\\-registry\\-snapshot|Mozilla|mwmbl|Neuronto|oauth4webapi|Orbit\\-MCP\\-Registry\\-IconResolver|PackageHound|packages\\.ecosyste\\.ms|PerplexityBot|personal\\-agent\\-platform\\-readonly\\-audit|pip|ProofBench|python\\-requests|QtCreator|reliability\\-bureau\\-spike|repology\\-linkchecker|rootz\\-mcp\\-registry\\-prober|SaSame\\-Census\\-Era\\-Probe|SaSame\\-MCP\\-Audit|SaSameAgentAudit|SentinelOracle|ShapBot|SmitheryBot|SolvedEarthPriceBot|spanly\\-enrich|strand\\-mcp|TAR\\-Directory\\-Indexer|TAR\\-Discovery|TAR\\-Health|Telegram|TelegramBot|teppi\\-probe|truespar\\-mcp\\-registry|trustoven\\-manifest\\-observer|utopian\\-foundry\\-probe|VerifyMCP\\-OwnersBot|Waggle|x402\\-observatory|YandexBot)",
 "observed_at_edge_note": "147 named clients observed requesting this host between 2026-08-31T20:58:11+00:00 and 2026-09-03T00:22:55+00:00, one page each under https://www.pathwren.workers.dev/bot/ and the whole set at https://www.pathwren.workers.dev/data/observed-clients.json. Observation, not a directory: it says these names arrived here in that window and nothing about what they are for."
}