{
  "botcentral": "1.1",
  "note": "Control tokens (Google-Extended, Applebot-Extended) are not HTTP User-Agents. They only exist in robots.txt.",
  "agents": [
    {
      "token": "GPTBot",
      "operator": "OpenAI",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://platform.openai.com/docs/bots",
      "notes": "Foundation-model training crawler. Distinct from ChatGPT search and user fetches.",
      "respectsRobots": true
    },
    {
      "token": "ChatGPT-User",
      "operator": "OpenAI",
      "purposes": [
        "retrieve",
        "act"
      ],
      "kind": "user-fetch",
      "docs": "https://platform.openai.com/docs/bots",
      "notes": "On-demand fetch when a person asks ChatGPT to read a page.",
      "respectsRobots": true
    },
    {
      "token": "OAI-SearchBot",
      "operator": "OpenAI",
      "purposes": [
        "search",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://platform.openai.com/docs/bots",
      "notes": "Search indexing for ChatGPT search. Not a training crawler.",
      "respectsRobots": true
    },
    {
      "token": "ClaudeBot",
      "operator": "Anthropic",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "notes": "Training crawler for Claude models.",
      "respectsRobots": true
    },
    {
      "token": "Claude-User",
      "operator": "Anthropic",
      "purposes": [
        "retrieve",
        "act"
      ],
      "kind": "user-fetch",
      "docs": "https://support.anthropic.com/",
      "notes": "User-initiated fetch from Claude.",
      "respectsRobots": true
    },
    {
      "token": "Claude-SearchBot",
      "operator": "Anthropic",
      "purposes": [
        "search",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://support.anthropic.com/",
      "notes": "Improves Claude search result quality.",
      "respectsRobots": true
    },
    {
      "token": "Google-Extended",
      "operator": "Google",
      "purposes": [
        "train"
      ],
      "kind": "control-token",
      "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "notes": "robots.txt control token, not a separate User-Agent. Governs Gemini training and grounding. Does not affect Google Search ranking.",
      "respectsRobots": true
    },
    {
      "token": "Googlebot",
      "operator": "Google",
      "purposes": [
        "search"
      ],
      "kind": "crawler",
      "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
      "notes": "Web search crawler. Not a Gemini training signal.",
      "respectsRobots": true
    },
    {
      "token": "Bingbot",
      "operator": "Microsoft",
      "purposes": [
        "search",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8cba48a4",
      "notes": "Bing and Copilot retrieval depend on Bing's index.",
      "respectsRobots": true
    },
    {
      "token": "Applebot",
      "operator": "Apple",
      "purposes": [
        "search"
      ],
      "kind": "crawler",
      "docs": "https://support.apple.com/en-us/119829",
      "notes": "Powers Spotlight, Siri suggestions, and Safari. Not Apple Intelligence training.",
      "respectsRobots": true
    },
    {
      "token": "Applebot-Extended",
      "operator": "Apple",
      "purposes": [
        "train"
      ],
      "kind": "control-token",
      "docs": "https://support.apple.com/en-us/119829",
      "notes": "Control token for Apple Intelligence training. Separate from Applebot search.",
      "respectsRobots": true
    },
    {
      "token": "PerplexityBot",
      "operator": "Perplexity",
      "purposes": [
        "search",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://docs.perplexity.ai/guides/bots",
      "notes": "Builds Perplexity's search index.",
      "respectsRobots": true
    },
    {
      "token": "Perplexity-User",
      "operator": "Perplexity",
      "purposes": [
        "retrieve",
        "act"
      ],
      "kind": "user-fetch",
      "docs": "https://docs.perplexity.ai/guides/bots",
      "notes": "User-initiated fetch from a Perplexity answer.",
      "respectsRobots": true
    },
    {
      "token": "Amazonbot",
      "operator": "Amazon",
      "purposes": [
        "retrieve",
        "search"
      ],
      "kind": "crawler",
      "docs": "https://developer.amazon.com/amazonbot",
      "notes": "Supports Alexa answers and Amazon services.",
      "respectsRobots": true
    },
    {
      "token": "Meta-ExternalAgent",
      "operator": "Meta",
      "purposes": [
        "train",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "notes": "AI training and ranking for Meta AI products.",
      "respectsRobots": true
    },
    {
      "token": "FacebookBot",
      "operator": "Meta",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "notes": "Documented as a training crawler.",
      "respectsRobots": true
    },
    {
      "token": "CCBot",
      "operator": "Common Crawl",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://commoncrawl.org/faq",
      "notes": "Open web corpus. Widely reused as training data by other labs.",
      "respectsRobots": true
    },
    {
      "token": "Bytespider",
      "operator": "ByteDance",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://bytespider.dev/",
      "notes": "TikTok / ByteDance model training crawler.",
      "respectsRobots": true
    },
    {
      "token": "cohere-ai",
      "operator": "Cohere",
      "purposes": [
        "retrieve"
      ],
      "kind": "user-fetch",
      "docs": "https://cohere.com/",
      "notes": "User-initiated retrieval for Cohere answers.",
      "respectsRobots": true
    },
    {
      "token": "cohere-training-data-crawler",
      "operator": "Cohere",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://cohere.com/",
      "notes": "Training-data crawler. Separate from cohere-ai retrieval.",
      "respectsRobots": true
    },
    {
      "token": "Diffbot",
      "operator": "Diffbot",
      "purposes": [
        "retrieve",
        "train"
      ],
      "kind": "crawler",
      "docs": "https://www.diffbot.com/",
      "notes": "Knowledge-graph extraction used by downstream AI products.",
      "respectsRobots": true
    },
    {
      "token": "DuckAssistBot",
      "operator": "DuckDuckGo",
      "purposes": [
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassist",
      "notes": "Fetches sources for Duck.ai answers.",
      "respectsRobots": true
    },
    {
      "token": "DeepSeekBot",
      "operator": "DeepSeek",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://www.deepseek.com/",
      "notes": "Training and product-improvement crawler.",
      "respectsRobots": true
    },
    {
      "token": "GrokBot",
      "operator": "xAI",
      "purposes": [
        "train"
      ],
      "kind": "crawler",
      "docs": "https://x.ai",
      "notes": "robots.txt token used to opt in or out of xAI training crawls.",
      "respectsRobots": true
    },
    {
      "token": "YouBot",
      "operator": "You.com",
      "purposes": [
        "search",
        "retrieve"
      ],
      "kind": "crawler",
      "docs": "https://you.com/",
      "notes": "You.com search and assistant retrieval.",
      "respectsRobots": true
    }
  ]
}
