{
  "$schema": "./ai-bots.schema.json",
  "name": "CodoSEO AI bot registry",
  "version": 1,
  "updated": "2026-10-09",
  "license": "CC0-1.0",
  "homepage": "https://codoseo.com/ai-bots",
  "bots": [
    {
      "token": "GPTBot",
      "operator": "OpenAI",
      "product": "GPTBot",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "GPTBot",
      "ip_ranges_url": "https://openai.com/gptbot.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.openai.com/api/docs/bots",
      "last_reviewed": "2026-10-09",
      "notes": "Used to train OpenAI's generative AI foundation models."
    },
    {
      "token": "OAI-SearchBot",
      "operator": "OpenAI",
      "product": "ChatGPT search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "OAI-SearchBot",
      "ip_ranges_url": "https://openai.com/searchbot.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.openai.com/api/docs/bots",
      "last_reviewed": "2026-10-09",
      "notes": "Blocking it removes the site from ChatGPT search answers; changes take about 24 hours."
    },
    {
      "token": "ChatGPT-User",
      "operator": "OpenAI",
      "product": "ChatGPT and Custom GPTs",
      "purpose": "user_fetch",
      "honours_robots": "partial",
      "crawls": true,
      "user_agent_contains": "ChatGPT-User",
      "ip_ranges_url": "https://openai.com/chatgpt-user.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.openai.com/api/docs/bots",
      "last_reviewed": "2026-10-09",
      "notes": "User-initiated, so robots.txt rules may not apply."
    },
    {
      "token": "OAI-AdsBot",
      "operator": "OpenAI",
      "product": "ChatGPT ads",
      "purpose": "ads",
      "honours_robots": "unknown",
      "crawls": true,
      "user_agent_contains": "OAI-AdsBot",
      "ip_ranges_url": "https://openai.com/adsbot.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.openai.com/api/docs/bots",
      "last_reviewed": "2026-10-09",
      "notes": "Checks pages submitted as ChatGPT ads; the documentation does not state its robots.txt behaviour."
    },
    {
      "token": "ClaudeBot",
      "operator": "Anthropic",
      "product": "Claude training",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "ClaudeBot",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "last_reviewed": "2026-10-09",
      "notes": "Collects web content for model training; also honours Crawl-delay."
    },
    {
      "token": "Claude-SearchBot",
      "operator": "Anthropic",
      "product": "Claude search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Claude-SearchBot",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "last_reviewed": "2026-10-09",
      "notes": "Improves Claude search result quality; blocking it can reduce visibility in Claude search."
    },
    {
      "token": "Claude-User",
      "operator": "Anthropic",
      "product": "Claude user fetch",
      "purpose": "user_fetch",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Claude-User",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "last_reviewed": "2026-10-09",
      "notes": "Fetches pages when a user asks Claude; honours robots.txt, unlike most user-triggered fetchers."
    },
    {
      "token": "PerplexityBot",
      "operator": "Perplexity",
      "product": "Perplexity search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "PerplexityBot",
      "ip_ranges_url": "https://www.perplexity.com/perplexitybot.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://docs.perplexity.ai/guides/bots",
      "last_reviewed": "2026-10-09",
      "notes": "Surfaces and links sites in Perplexity search results."
    },
    {
      "token": "Perplexity-User",
      "operator": "Perplexity",
      "product": "Perplexity user fetch",
      "purpose": "user_fetch",
      "honours_robots": "no",
      "crawls": true,
      "user_agent_contains": "Perplexity-User",
      "ip_ranges_url": "https://www.perplexity.com/perplexity-user.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://docs.perplexity.ai/guides/bots",
      "last_reviewed": "2026-10-09",
      "notes": "Since a user requested the fetch, this fetcher generally ignores robots.txt rules."
    },
    {
      "token": "Googlebot",
      "operator": "Google",
      "product": "Google Search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Googlebot",
      "ip_ranges_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "reverse_dns": [
        "googlebot.com",
        "google.com",
        "googleusercontent.com"
      ],
      "signature_agent": null,
      "source_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "last_reviewed": "2026-10-09",
      "notes": "Google's search crawler."
    },
    {
      "token": "Google-Extended",
      "operator": "Google",
      "product": "Gemini training and grounding",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": false,
      "user_agent_contains": null,
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "last_reviewed": "2026-10-09",
      "notes": "Control token only, with no user agent of its own; governs Gemini training and grounding and does not affect Search or AI Overviews."
    },
    {
      "token": "Google-Agent",
      "operator": "Google",
      "product": "Project Mariner agents",
      "purpose": "agent",
      "honours_robots": "no",
      "crawls": true,
      "user_agent_contains": "Google-Agent",
      "ip_ranges_url": "https://developers.google.com/static/crawling/ipranges/user-triggered-agents.json",
      "reverse_dns": [
        "google.com",
        "googleusercontent.com"
      ],
      "signature_agent": "https://agent.bot.goog",
      "source_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-agent",
      "last_reviewed": "2026-10-09",
      "notes": "User-triggered agent on Google infrastructure; generally ignores robots.txt and is verified by IP list or Web Bot Auth."
    },
    {
      "token": "Applebot",
      "operator": "Apple",
      "product": "Spotlight, Siri and Safari search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Applebot",
      "ip_ranges_url": "https://search.developer.apple.com/applebot.json",
      "reverse_dns": [
        "applebot.apple.com"
      ],
      "signature_agent": null,
      "source_url": "https://support.apple.com/en-us/119829",
      "last_reviewed": "2026-10-09",
      "notes": "Crawls for Apple search features; its data may also be used to train Apple foundation models.",
      "robots_fallback": "Googlebot"
    },
    {
      "token": "Applebot-Extended",
      "operator": "Apple",
      "product": "Apple foundation model training",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": false,
      "user_agent_contains": null,
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://support.apple.com/en-us/119829",
      "last_reviewed": "2026-10-09",
      "notes": "Does not crawl; a usage-permission token for Apple model training only."
    },
    {
      "token": "meta-externalagent",
      "operator": "Meta",
      "product": "Meta AI model training",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "meta-externalagent",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "last_reviewed": "2026-10-09",
      "notes": "Collects content to help build Meta's AI models; no IP list published on the page."
    },
    {
      "token": "meta-webindexer",
      "operator": "Meta",
      "product": "Meta AI search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "meta-webindexer",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "last_reviewed": "2026-10-09",
      "notes": "Improves Meta AI search results; allowing it helps Meta cite and link to the site."
    },
    {
      "token": "meta-externalfetcher",
      "operator": "Meta",
      "product": "Meta AI user fetch",
      "purpose": "user_fetch",
      "honours_robots": "partial",
      "crawls": true,
      "user_agent_contains": "meta-externalfetcher",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "last_reviewed": "2026-10-09",
      "notes": "Fetches links at a user's request and may bypass robots.txt rules."
    },
    {
      "token": "meta-externalads",
      "operator": "Meta",
      "product": "Meta ads",
      "purpose": "ads",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "meta-externalads",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "last_reviewed": "2026-10-07",
      "notes": "Crawls for advertising and other business products."
    },
    {
      "token": "Amazonbot",
      "operator": "Amazon",
      "product": "Amazonbot",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Amazonbot",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developer.amazon.com/amazonbot",
      "last_reviewed": "2026-10-09",
      "notes": "Used to improve Amazon products and for AI model training; IP list is an HTML page, not JSON."
    },
    {
      "token": "Amzn-SearchBot",
      "operator": "Amazon",
      "product": "Alexa and Amazon search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "Amzn-SearchBot",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developer.amazon.com/amazonbot",
      "last_reviewed": "2026-10-09",
      "notes": "Improves search in Alexa and other Amazon products; not used for generative AI training; IP list is an HTML page, not JSON."
    },
    {
      "token": "Amzn-User",
      "operator": "Amazon",
      "product": "Amazon user fetch",
      "purpose": "user_fetch",
      "honours_robots": "partial",
      "crawls": true,
      "user_agent_contains": "Amzn-User",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://developer.amazon.com/amazonbot",
      "last_reviewed": "2026-10-09",
      "notes": "Fetches pages for live user queries and may not follow all robots.txt directives; IP list is an HTML page, not JSON."
    },
    {
      "token": "Bingbot",
      "operator": "Microsoft",
      "product": "Bing and Copilot search",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "bingbot",
      "ip_ranges_url": "https://www.bing.com/toolbox/bingbot.json",
      "reverse_dns": [
        "search.msn.com"
      ],
      "signature_agent": null,
      "source_url": "https://www.bing.com/toolbox/verify-bingbot",
      "last_reviewed": "2026-10-07",
      "notes": "Microsoft controls Copilot use of content with page-level meta tags, not a separate robots token."
    },
    {
      "token": "DuckAssistBot",
      "operator": "DuckDuckGo",
      "product": "DuckDuckGo AI-assisted answers",
      "purpose": "search",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "DuckAssistBot",
      "ip_ranges_url": "https://duckduckgo.com/duckassistbot.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
      "last_reviewed": "2026-10-09",
      "notes": "Crawls in real time for AI-assisted answers and is not used for training; a robots.txt opt-out takes about 72 hours."
    },
    {
      "token": "MistralAI-Index",
      "operator": "Mistral",
      "product": "Mistral search index",
      "purpose": "search",
      "honours_robots": "unknown",
      "crawls": true,
      "user_agent_contains": "MistralAI-Index",
      "ip_ranges_url": "https://mistral.ai/mistralai-index-ips.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://docs.mistral.ai/robots",
      "last_reviewed": "2026-10-09",
      "notes": "Indexes pages to answer questions in Mistral products; the documentation does not state robots.txt behaviour."
    },
    {
      "token": "MistralAI-User",
      "operator": "Mistral",
      "product": "Mistral user fetch",
      "purpose": "user_fetch",
      "honours_robots": "unknown",
      "crawls": true,
      "user_agent_contains": "MistralAI-User",
      "ip_ranges_url": "https://mistral.ai/mistralai-user-ips.json",
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://docs.mistral.ai/robots",
      "last_reviewed": "2026-10-09",
      "notes": "Visits pages when a user asks a question; the documentation does not state robots.txt behaviour."
    },
    {
      "token": "MistralAI-Training",
      "operator": "Mistral",
      "product": "Mistral model training",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "MistralAI-Training",
      "ip_ranges_url": null,
      "reverse_dns": [],
      "signature_agent": null,
      "source_url": "https://docs.mistral.ai/robots",
      "last_reviewed": "2026-10-09",
      "notes": "Builds training datasets; webmasters can disallow it in robots.txt. No IP list is published."
    },
    {
      "token": "CCBot",
      "operator": "Common Crawl",
      "product": "Common Crawl",
      "purpose": "training",
      "honours_robots": "yes",
      "crawls": true,
      "user_agent_contains": "CCBot",
      "ip_ranges_url": "https://index.commoncrawl.org/ccbot.json",
      "reverse_dns": [
        "crawl.commoncrawl.org"
      ],
      "signature_agent": null,
      "source_url": "https://commoncrawl.org/ccbot",
      "last_reviewed": "2026-10-09",
      "notes": "Builds the open Common Crawl corpus used to train many models; spoofing is common, so verify by IP."
    }
  ]
}
