{
 "operator": "Webz.io",
 "slug": "webz",
 "docs": "https://webz.io/blog/machine-learning/",
 "crawler_count": 3,
 "crawlers": [
  {
   "slug": "omgili",
   "name": "omgili",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "omgili",
   "user_agent_substring": "omgili",
   "user_agent_example": "omgili/0.5 +http://omgili.com",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The older robots token for the same Webz.io collection, still honoured and still worth listing.",
   "cost_of_blocking": "Same as omgilibot.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/omgili.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/omgili.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "omgilibot",
   "name": "omgilibot",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "omgilibot",
   "user_agent_substring": "omgilibot",
   "user_agent_example": "omgilibot/0.4; +http://omgili.com",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Webz.io's crawler, collecting web and forum text sold as datasets, including to model builders.",
   "cost_of_blocking": "Exclusion from a commercial dataset resold to third parties.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/omgilibot.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/omgilibot.json",
   "last_reviewed": "2026-09-01"
  },
  {
   "slug": "webzio-extended",
   "name": "Webzio-Extended",
   "operator": "Webz.io",
   "operator_slug": "webz",
   "category": "ai-training",
   "category_label": "AI training crawlers",
   "robots_token": "Webzio-Extended",
   "user_agent_substring": "Webzio-Extended",
   "user_agent_example": "Mozilla/5.0 (compatible; Webzio-Extended/1.0)",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "Webz.io's opt-out token specifically for AI training reuse, in the pattern Google and Apple established.",
   "cost_of_blocking": "Your content is excluded from the AI-training tier of Webz.io's product while ordinary collection continues.",
   "operator_docs": "https://webz.io/blog/machine-learning/",
   "html_url": "https://www.pathwren.workers.dev/crawler/webzio-extended.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/webzio-extended.json",
   "last_reviewed": "2026-09-01"
  }
 ],
 "ip_range_endpoints": []
}