[
  { "name": "GPTBot", "operator": "OpenAI", "purpose": "AI model training", "docs": "https://platform.openai.com/docs/gptbot" },
  { "name": "ChatGPT-User", "operator": "OpenAI", "purpose": "ChatGPT browsing/plugins on behalf of a user", "docs": "https://platform.openai.com/docs/plugins/bot" },
  { "name": "OAI-SearchBot", "operator": "OpenAI", "purpose": "Powers ChatGPT search results", "docs": "https://platform.openai.com/docs/bots" },
  { "name": "ClaudeBot", "operator": "Anthropic", "purpose": "AI model training and web crawling", "docs": "https://support.anthropic.com/en/articles/8896518" },
  { "name": "Claude-User", "operator": "Anthropic", "purpose": "Claude fetching pages on behalf of a user (formerly Claude-Web)", "docs": "https://support.anthropic.com/en/articles/8896518" },
  { "name": "Claude-SearchBot", "operator": "Anthropic", "purpose": "Improves Claude's search result quality and relevance", "docs": "https://support.anthropic.com/en/articles/8896518" },
  { "name": "anthropic-ai", "operator": "Anthropic", "purpose": "AI model training", "docs": "https://support.anthropic.com/en/articles/8896518" },
  { "name": "PerplexityBot", "operator": "Perplexity", "purpose": "Powers Perplexity answer engine", "docs": "https://docs.perplexity.ai/guides/bots" },
  { "name": "Perplexity-User", "operator": "Perplexity", "purpose": "Perplexity fetching pages on behalf of a user", "docs": "https://docs.perplexity.ai/guides/bots" },
  { "name": "Google-Extended", "operator": "Google", "purpose": "Controls use of content for Gemini/Vertex AI training", "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers" },
  { "name": "GoogleOther", "operator": "Google", "purpose": "Internal research and development crawling", "docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers" },
  { "name": "Google-CloudVertexBot", "operator": "Google", "purpose": "Site-owner-requested crawls for building Vertex AI Agents", "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers" },
  { "name": "DuckAssistBot", "operator": "DuckDuckGo", "purpose": "Powers DuckDuckGo's AI-assisted answers (not used for model training)", "docs": "https://duckduckgo.com/duckassistbot.html" },
  { "name": "CCBot", "operator": "Common Crawl", "purpose": "Open web corpus used to train many AI models", "docs": "https://commoncrawl.org/ccbot" },
  { "name": "Bytespider", "operator": "ByteDance", "purpose": "AI model training (TikTok/Doubao)", "docs": "" },
  { "name": "Amazonbot", "operator": "Amazon", "purpose": "Alexa and AI product improvement", "docs": "https://developer.amazon.com/amazonbot" },
  { "name": "Applebot-Extended", "operator": "Apple", "purpose": "Controls use of content for Apple Intelligence training", "docs": "https://support.apple.com/en-us/119829" },
  { "name": "Meta-ExternalAgent", "operator": "Meta", "purpose": "AI model training and indexing", "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers" },
  { "name": "Meta-ExternalFetcher", "operator": "Meta", "purpose": "Fetches individual links at a user's request for agentic AI features", "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers" },
  { "name": "Meta-WebIndexer", "operator": "Meta", "purpose": "Improves Meta AI search result quality and relevance", "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers" },
  { "name": "FacebookBot", "operator": "Meta", "purpose": "Language model training crawler", "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers" },
  { "name": "cohere-ai", "operator": "Cohere", "purpose": "AI model training", "docs": "" },
  { "name": "Diffbot", "operator": "Diffbot", "purpose": "Structured web data extraction for AI/ML", "docs": "https://docs.diffbot.com/docs/crawling" },
  { "name": "ImagesiftBot", "operator": "ImageSift", "purpose": "Image dataset collection for AI training", "docs": "" },
  { "name": "Timpibot", "operator": "Timpi", "purpose": "Decentralized search index crawling", "docs": "" },
  { "name": "YouBot", "operator": "You.com", "purpose": "Powers You.com AI search", "docs": "" },
  { "name": "Ai2Bot", "operator": "Allen Institute for AI", "purpose": "Open dataset and model research", "docs": "" },
  { "name": "ChatGPT Agent", "operator": "OpenAI", "purpose": "Agentic browsing - navigates and interacts with sites to complete multi-step tasks for ChatGPT users", "docs": "https://developers.openai.com/api/docs/bots" },
  { "name": "DeepSeekBot", "operator": "DeepSeek", "purpose": "AI model training and product improvement", "docs": "" },
  { "name": "PetalBot", "operator": "Huawei", "purpose": "Powers Petal Search and AI recommendations in Huawei Assistant", "docs": "https://aspiegel.com/petalbot" },
  { "name": "omgili", "operator": "Webz.io", "purpose": "Crawls and resells web data, including for LLM training", "docs": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/" },
  { "name": "SemrushBot-OCOB", "operator": "Semrush", "purpose": "Crawls sites for Semrush's ContentShake AI tool", "docs": "https://www.semrush.com/bot/" }
]
