{
  "schema_version": 1,
  "name": "DeployRadar AI crawler registry",
  "description": "AI crawler robots.txt tokens, who operates each one, what it is for, whether it obeys robots.txt, and what blocking it costs. Every entry is sourced from the operator's documentation where one exists and carries the date it was last checked.",
  "url": "https://deployradar.dev/crawlers.json",
  "documentation": "https://deployradar.dev/crawlers",
  "version": "2026-09-19",
  "fields": {
    "token": "The robots.txt User-agent token, matched case-insensitively.",
    "operator": "The company that runs the crawler.",
    "purpose": "training | search_citation | user_fetch | mixed | corpus | info_only",
    "robots_token_only": "True when no crawler sends this user agent: the token only means something inside robots.txt.",
    "respects_robots": "yes | generally | may_bypass | unreliable | no | n/a, per the operator's documentation.",
    "user_agent": "A representative User-Agent header, or null for a robots-only token.",
    "ip_ranges_url": "The operator's published IP ranges, or null where it publishes none.",
    "docs_url": "The operator's documentation, or the best available source where there is none.",
    "if_blocked": "What blocking the token costs a site.",
    "last_verified": "The day this entry was last checked against its source.",
    "page": "The entry's page on this site."
  },
  "crawlers": [
    {
      "token": "GPTBot",
      "operator": "OpenAI",
      "purpose": "training",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
      "ip_ranges_url": "https://openai.com/gptbot.json",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "if_blocked": "Excluded from future OpenAI model training. Does NOT affect ChatGPT search citations (that's OAI-SearchBot).",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/gptbot"
    },
    {
      "token": "OAI-SearchBot",
      "operator": "OpenAI",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
      "ip_ranges_url": "https://openai.com/searchbot.json",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "if_blocked": "Your site disappears from ChatGPT search answers and citations. The most damaging accidental block we see.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/oai-searchbot"
    },
    {
      "token": "ChatGPT-User",
      "operator": "OpenAI",
      "purpose": "user_fetch",
      "robots_token_only": false,
      "respects_robots": "may_bypass",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
      "ip_ranges_url": "https://openai.com/chatgpt-user.json",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "if_blocked": "Signals opt-out of live page fetches when ChatGPT users ask about your site, but OpenAI notes robots.txt rules 'may not apply' to user-initiated fetches.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/chatgpt-user"
    },
    {
      "token": "OAI-AdsBot",
      "operator": "OpenAI",
      "purpose": "info_only",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot",
      "ip_ranges_url": "https://openai.com/adsbot.json",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "if_blocked": "Landing pages you submit as ChatGPT ads cannot be checked, so the ad is not approved. No effect on organic ChatGPT visibility. That is OAI-SearchBot.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/oai-adsbot"
    },
    {
      "token": "ClaudeBot",
      "operator": "Anthropic",
      "purpose": "training",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518",
      "if_blocked": "Excluded from future Anthropic model training. Does not affect Claude's search visibility (that's Claude-SearchBot).",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/claudebot"
    },
    {
      "token": "Claude-SearchBot",
      "operator": "Anthropic",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-SearchBot/1.0; +https://claude.com)",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518",
      "if_blocked": "Reduced visibility in Claude's search results.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/claude-searchbot"
    },
    {
      "token": "Claude-User",
      "operator": "Anthropic",
      "purpose": "user_fetch",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-User/1.0; +https://claude.com)",
      "ip_ranges_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518",
      "if_blocked": "Claude cannot fetch your pages when users ask about your site or product.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/claude-user"
    },
    {
      "token": "PerplexityBot",
      "operator": "Perplexity",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "generally",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
      "ip_ranges_url": "https://www.perplexity.com/perplexitybot.json",
      "docs_url": "https://docs.perplexity.ai/guides/bots",
      "if_blocked": "Your site will not be surfaced or linked in Perplexity answers.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/perplexitybot"
    },
    {
      "token": "Perplexity-User",
      "operator": "Perplexity",
      "purpose": "user_fetch",
      "robots_token_only": false,
      "respects_robots": "may_bypass",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)",
      "ip_ranges_url": "https://www.perplexity.com/perplexity-user.json",
      "docs_url": "https://docs.perplexity.ai/guides/bots",
      "if_blocked": "Advisory only. Perplexity's own docs state this user-triggered fetcher 'generally ignores robots.txt rules'.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/perplexity-user"
    },
    {
      "token": "Google-Extended",
      "operator": "Google",
      "purpose": "training",
      "robots_token_only": true,
      "respects_robots": "yes",
      "user_agent": null,
      "ip_ranges_url": null,
      "docs_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
      "if_blocked": "Blocks Gemini model training and grounding (content fed from the Search index to Gemini at prompt time). DOES NOT remove you from AI Overviews or AI Mode, and does not affect Search ranking. If you blocked this to hide from AI Overviews, it isn't doing what you think.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/google-extended"
    },
    {
      "token": "Googlebot",
      "operator": "Google",
      "purpose": "info_only",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": null,
      "ip_ranges_url": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
      "docs_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
      "if_blocked": "Removes you from Google Search entirely, including AI Overviews and AI Mode, which ride ordinary Search crawling.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/googlebot"
    },
    {
      "token": "Applebot",
      "operator": "Apple",
      "purpose": "mixed",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": null,
      "ip_ranges_url": "https://search.developer.apple.com/applebot.json",
      "docs_url": "https://support.apple.com/en-us/119829",
      "if_blocked": "Removed from Siri, Spotlight, Safari search, and Apple Intelligence citations.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/applebot"
    },
    {
      "token": "Applebot-Extended",
      "operator": "Apple",
      "purpose": "training",
      "robots_token_only": true,
      "respects_robots": "yes",
      "user_agent": null,
      "ip_ranges_url": null,
      "docs_url": "https://support.apple.com/en-us/119829",
      "if_blocked": "Content excluded from Apple foundation-model training. Siri/Spotlight presence unaffected.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/applebot-extended"
    },
    {
      "token": "meta-externalagent",
      "operator": "Meta",
      "purpose": "training",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)",
      "ip_ranges_url": null,
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "if_blocked": "Content excluded from Meta AI training and direct indexing.",
      "last_verified": "2026-09-19",
      "page": "https://deployradar.dev/crawlers/meta-externalagent"
    },
    {
      "token": "meta-webindexer",
      "operator": "Meta",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "meta-webindexer/1.1 (+https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers)",
      "ip_ranges_url": null,
      "docs_url": "https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers",
      "if_blocked": "Your site will not be cited or linked in Meta AI answers.",
      "last_verified": "2026-09-15",
      "page": "https://deployradar.dev/crawlers/meta-webindexer"
    },
    {
      "token": "meta-externalfetcher",
      "operator": "Meta",
      "purpose": "user_fetch",
      "robots_token_only": false,
      "respects_robots": "may_bypass",
      "user_agent": "meta-externalfetcher/1.1 (+https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers)",
      "ip_ranges_url": null,
      "docs_url": "https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers",
      "if_blocked": "Advisory only. Meta's own documentation says this fetcher may bypass robots.txt, because the fetch is something a user asked for.",
      "last_verified": "2026-09-19",
      "page": "https://deployradar.dev/crawlers/meta-externalfetcher"
    },
    {
      "token": "meta-externalads",
      "operator": "Meta",
      "purpose": "info_only",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "meta-externalads/1.1 (+https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers)",
      "ip_ranges_url": null,
      "docs_url": "https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers",
      "if_blocked": "Meta cannot read the landing pages behind your own ads, which degrades ad relevance and business products you are paying for. No effect on organic Meta AI visibility. That is meta-webindexer.",
      "last_verified": "2026-09-19",
      "page": "https://deployradar.dev/crawlers/meta-externalads"
    },
    {
      "token": "facebookexternalhit",
      "operator": "Meta",
      "purpose": "info_only",
      "robots_token_only": false,
      "respects_robots": "generally",
      "user_agent": "facebookexternalhit/1.1 (+http://www.facebook.com/externalhit_uatext.php)",
      "ip_ranges_url": null,
      "docs_url": "https://developers.facebook.com/documentation/sharing/webmasters/web-crawlers",
      "if_blocked": "Link previews stop working. A link to your site shared on Facebook, Instagram, WhatsApp or Messenger renders as a bare URL with no title, description or image.",
      "last_verified": "2026-09-19",
      "page": "https://deployradar.dev/crawlers/facebookexternalhit"
    },
    {
      "token": "Amazonbot",
      "operator": "Amazon",
      "purpose": "mixed",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot) Chrome/119.0.6045.214 Safari/537.36",
      "ip_ranges_url": "https://developer.amazon.com/amazonbot/ip-addresses/",
      "docs_url": "https://developer.amazon.com/amazonbot",
      "if_blocked": "Out of Alexa answers AND Amazon AI training, because one token controls both; there is no way to split them.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/amazonbot"
    },
    {
      "token": "Bytespider",
      "operator": "ByteDance",
      "purpose": "training",
      "robots_token_only": false,
      "respects_robots": "unreliable",
      "user_agent": "Mozilla/5.0 (Linux; Android 5.0) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; Bytespider; spider-feedback@bytedance.com)",
      "ip_ranges_url": null,
      "docs_url": null,
      "if_blocked": "Best-effort only. ByteDance publishes no official crawler documentation or IP ranges, and Bytespider is widely reported to ignore robots.txt.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/bytespider"
    },
    {
      "token": "CCBot",
      "operator": "Common Crawl",
      "purpose": "corpus",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "CCBot/2.0 (https://commoncrawl.org/faq/)",
      "ip_ranges_url": "https://index.commoncrawl.org/ccbot.json",
      "docs_url": "https://commoncrawl.org/ccbot",
      "if_blocked": "Excluded from future Common Crawl snapshots, the open corpus many AI labs train on. Indirectly reduces presence in many models' training data (not retroactive).",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/ccbot"
    },
    {
      "token": "MistralAI-Training",
      "operator": "Mistral",
      "purpose": "training",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Training/1.0; +https://docs.mistral.ai/robots)",
      "ip_ranges_url": null,
      "docs_url": "https://docs.mistral.ai/robots",
      "if_blocked": "Out of Mistral model training.",
      "last_verified": "2026-09-15",
      "page": "https://deployradar.dev/crawlers/mistralai-training"
    },
    {
      "token": "MistralAI-Index",
      "operator": "Mistral",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Index/1.0; +https://docs.mistral.ai/robots)",
      "ip_ranges_url": "https://mistral.ai/mistralai-index-ips.json",
      "docs_url": "https://docs.mistral.ai/robots",
      "if_blocked": "Out of Le Chat's search results and citations.",
      "last_verified": "2026-09-15",
      "page": "https://deployradar.dev/crawlers/mistralai-index"
    },
    {
      "token": "MistralAI-User",
      "operator": "Mistral",
      "purpose": "user_fetch",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-User/1.0; +https://docs.mistral.ai/robots)",
      "ip_ranges_url": "https://mistral.ai/mistralai-user-ips.json",
      "docs_url": "https://docs.mistral.ai/robots",
      "if_blocked": "Le Chat cannot open your pages when somebody asks it about them.",
      "last_verified": "2026-09-15",
      "page": "https://deployradar.dev/crawlers/mistralai-user"
    },
    {
      "token": "DuckAssistBot",
      "operator": "DuckDuckGo",
      "purpose": "search_citation",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": "Mozilla/5.0 (compatible; DuckAssistBot/1.2; +http://duckduckgo.com/duckassistbot.html)",
      "ip_ranges_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
      "docs_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
      "if_blocked": "Out of DuckAssist AI answers.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/duckassistbot"
    },
    {
      "token": "Bingbot",
      "operator": "Microsoft",
      "purpose": "info_only",
      "robots_token_only": false,
      "respects_robots": "yes",
      "user_agent": null,
      "ip_ranges_url": "https://www.bing.com/toolbox/bingbot.json",
      "docs_url": "https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat",
      "if_blocked": "Removes you from Bing Search AND Copilot answers together, and there is no separate Microsoft AI crawler token.",
      "last_verified": "2026-09-02",
      "page": "https://deployradar.dev/crawlers/bingbot"
    }
  ],
  "no_token": [
    {
      "operator": "xAI (Grok)",
      "note": "xAI publishes no official crawler identity. Directory-listed tokens (GrokBot, xAI-Grok) have never been observed in real traffic; Grok fetches with spoofed browser user agents from datacenter IPs. No robots.txt rule can control it, and anyone telling you otherwise is guessing.",
      "docs_url": "https://stackfox.co/research/grok-user-agent",
      "last_verified": "2026-09-02"
    }
  ]
}
