{
  "name": "canaicrawl bot list",
  "url": "https://canaicrawl.com/bots",
  "source": "https://canaicrawl.com/bots.json",
  "license": "Free to use; please link back to canaicrawl.com.",
  "updated": "2026-09-30",
  "groups": [
    {
      "key": "search",
      "label": "Search engines",
      "purpose": "Search index",
      "weight": 40,
      "description": "The crawlers behind Google, Bing and the other search engines. Blocking one drops you out of its results."
    },
    {
      "key": "aisearch",
      "label": "AI search",
      "purpose": "AI answers",
      "weight": 25,
      "description": "Crawlers that build the indexes AI assistants answer from and cite. Blocking one keeps you out of its answers."
    },
    {
      "key": "training",
      "label": "AI training",
      "purpose": "Model training",
      "weight": 10,
      "description": "Crawlers that collect pages to train models. Blocking them does not affect search or AI answers."
    },
    {
      "key": "fetch",
      "label": "User-triggered fetchers",
      "purpose": "Live user request",
      "weight": 10,
      "description": "Fetchers that load a page because a person asked an assistant about it. Blocking one means the assistant can’t open your pages on request."
    },
    {
      "key": "seo",
      "label": "SEO tools",
      "purpose": "Backlink & SEO index",
      "weight": 5,
      "description": "Crawlers behind backlink and SEO tools. Many sites block them to save bandwidth; it has no effect on search."
    }
  ],
  "bots": [
    {
      "name": "Googlebot",
      "slug": "googlebot",
      "vendor": "Google",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "Googlebot",
      "userAgent": "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
      "tokenOnly": false,
      "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
      "about": "Google’s main web crawler. It builds the index behind Google Search, and its fetches also feed AI Overviews unless Google-Extended is disallowed.",
      "url": "https://canaicrawl.com/bots/googlebot"
    },
    {
      "name": "Bingbot",
      "slug": "bingbot",
      "vendor": "Microsoft",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "Bingbot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) Chrome/131.0.0.0 Safari/537.36",
      "tokenOnly": false,
      "docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "about": "Microsoft’s crawler for Bing. Its index also powers Copilot answers, DuckDuckGo results and Yahoo Search.",
      "url": "https://canaicrawl.com/bots/bingbot"
    },
    {
      "name": "DuckDuckBot",
      "slug": "duckduckbot",
      "vendor": "DuckDuckGo",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "DuckDuckBot",
      "userAgent": "DuckDuckBot/1.1; (+http://duckduckgo.com/duckduckbot.html)",
      "tokenOnly": false,
      "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot",
      "about": "DuckDuckGo’s crawler. It supplements the Bing index that most DuckDuckGo results come from.",
      "url": "https://canaicrawl.com/bots/duckduckbot"
    },
    {
      "name": "Applebot",
      "slug": "applebot",
      "vendor": "Apple",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "Applebot",
      "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
      "tokenOnly": false,
      "docs": "https://support.apple.com/en-us/119829",
      "about": "Apple’s crawler for Siri and Spotlight suggestions. When robots.txt doesn’t name it, it follows the Googlebot rules.",
      "url": "https://canaicrawl.com/bots/applebot"
    },
    {
      "name": "PetalBot",
      "slug": "petalbot",
      "vendor": "Huawei",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "PetalBot",
      "userAgent": "Mozilla/5.0 (Linux; Android 7.0;) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; PetalBot;+https://webmaster.petalsearch.com/site/petalbot)",
      "tokenOnly": false,
      "docs": "https://webmaster.petalsearch.com/site/petalbot",
      "about": "Huawei’s crawler for Petal Search, the default search on Huawei phones outside China.",
      "url": "https://canaicrawl.com/bots/petalbot"
    },
    {
      "name": "YandexBot",
      "slug": "yandexbot",
      "vendor": "Yandex",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "YandexBot",
      "userAgent": "Mozilla/5.0 (compatible; YandexBot/3.0; +http://yandex.com/bots)",
      "tokenOnly": false,
      "docs": "https://yandex.com/support/webmaster/robot-workings/check-yandex-robots.html",
      "about": "The main crawler of Yandex, the largest search engine in Russia.",
      "url": "https://canaicrawl.com/bots/yandexbot"
    },
    {
      "name": "Baiduspider",
      "slug": "baiduspider",
      "vendor": "Baidu",
      "group": "search",
      "purpose": "Search index",
      "robotsToken": "Baiduspider",
      "userAgent": "Mozilla/5.0 (compatible; Baiduspider/2.0; +http://www.baidu.com/search/spider.html)",
      "tokenOnly": false,
      "docs": "https://www.baidu.com/search/spider.html",
      "about": "The crawler of Baidu, the largest search engine in China.",
      "url": "https://canaicrawl.com/bots/baiduspider"
    },
    {
      "name": "OAI-SearchBot",
      "slug": "oai-searchbot",
      "vendor": "OpenAI",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "OAI-SearchBot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-SearchBot/1.0; +https://openai.com/searchbot",
      "tokenOnly": false,
      "docs": "https://platform.openai.com/docs/bots",
      "about": "Builds the index ChatGPT search answers from and links to. It is not used for training; that is GPTBot.",
      "url": "https://canaicrawl.com/bots/oai-searchbot"
    },
    {
      "name": "PerplexityBot",
      "slug": "perplexitybot",
      "vendor": "Perplexity",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "PerplexityBot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
      "tokenOnly": false,
      "docs": "https://docs.perplexity.ai/guides/bots",
      "about": "Indexes pages so Perplexity can cite them in answers. Perplexity says it is not used for training.",
      "url": "https://canaicrawl.com/bots/perplexitybot"
    },
    {
      "name": "Claude-SearchBot",
      "slug": "claude-searchbot",
      "vendor": "Anthropic",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "Claude-SearchBot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-SearchBot/1.0; +https://www.anthropic.com)",
      "tokenOnly": false,
      "docs": "https://support.claude.com/en/articles/8896518",
      "about": "Indexes pages to improve the quality of Claude’s web search results. Not used for training; that is ClaudeBot.",
      "url": "https://canaicrawl.com/bots/claude-searchbot"
    },
    {
      "name": "DuckAssistBot",
      "slug": "duckassistbot",
      "vendor": "DuckDuckGo",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "DuckAssistBot",
      "userAgent": "DuckAssistBot/1.0; (+http://duckduckgo.com/duckassistbot.html)",
      "tokenOnly": false,
      "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
      "about": "Fetches pages for DuckAssist, the AI-generated answers at the top of DuckDuckGo results.",
      "url": "https://canaicrawl.com/bots/duckassistbot"
    },
    {
      "name": "Amazonbot",
      "slug": "amazonbot",
      "vendor": "Amazon",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "Amazonbot",
      "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_10_1) AppleWebKit/600.2.5 (KHTML, like Gecko) Version/8.0.2 Safari/600.2.5 (Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot)",
      "tokenOnly": false,
      "docs": "https://developer.amazon.com/amazonbot",
      "about": "Amazon’s crawler, used so Alexa and other Amazon services can answer questions from web pages.",
      "url": "https://canaicrawl.com/bots/amazonbot"
    },
    {
      "name": "YouBot",
      "slug": "youbot",
      "vendor": "You.com",
      "group": "aisearch",
      "purpose": "AI answers",
      "robotsToken": "YouBot",
      "userAgent": "Mozilla/5.0 (compatible; YouBot (+http://www.you.com))",
      "tokenOnly": false,
      "docs": "https://about.you.com/youbot/",
      "about": "The crawler behind You.com’s AI search and answers.",
      "url": "https://canaicrawl.com/bots/youbot"
    },
    {
      "name": "GPTBot",
      "slug": "gptbot",
      "vendor": "OpenAI",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "GPTBot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.1; +https://openai.com/gptbot",
      "tokenOnly": false,
      "docs": "https://platform.openai.com/docs/bots",
      "about": "Collects pages that may be used to train OpenAI’s models. Blocking it does not affect ChatGPT search, which uses OAI-SearchBot.",
      "url": "https://canaicrawl.com/bots/gptbot"
    },
    {
      "name": "ClaudeBot",
      "slug": "claudebot",
      "vendor": "Anthropic",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "ClaudeBot",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
      "tokenOnly": false,
      "docs": "https://support.claude.com/en/articles/8896518",
      "about": "Collects pages that may be used to train Anthropic’s models. Blocking it does not affect Claude’s search or user requests.",
      "url": "https://canaicrawl.com/bots/claudebot"
    },
    {
      "name": "Google-Extended",
      "slug": "google-extended",
      "vendor": "Google",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "Google-Extended",
      "userAgent": null,
      "tokenOnly": true,
      "docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-extended",
      "about": "Not a crawler. A robots.txt token that tells Google whether pages Googlebot fetched may train Gemini and be used for grounding. Disallowing it does not affect Google Search.",
      "url": "https://canaicrawl.com/bots/google-extended"
    },
    {
      "name": "CCBot",
      "slug": "ccbot",
      "vendor": "Common Crawl",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "CCBot",
      "userAgent": "CCBot/2.0 (https://commoncrawl.org/faq/)",
      "tokenOnly": false,
      "docs": "https://commoncrawl.org/ccbot",
      "about": "Builds the free Common Crawl archive, which many AI models have been trained on.",
      "url": "https://canaicrawl.com/bots/ccbot"
    },
    {
      "name": "Applebot-Extended",
      "slug": "applebot-extended",
      "vendor": "Apple",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "Applebot-Extended",
      "userAgent": null,
      "tokenOnly": true,
      "docs": "https://support.apple.com/en-us/119829",
      "about": "Not a crawler. A robots.txt token that tells Apple whether pages Applebot fetched may train Apple’s models. Disallowing it does not affect Siri or Spotlight.",
      "url": "https://canaicrawl.com/bots/applebot-extended"
    },
    {
      "name": "GoogleOther",
      "slug": "googleother",
      "vendor": "Google",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "GoogleOther",
      "userAgent": "GoogleOther",
      "tokenOnly": false,
      "docs": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#googleother",
      "about": "Google’s generic crawler for research and development crawls by product teams. It is not used for Google Search.",
      "url": "https://canaicrawl.com/bots/googleother"
    },
    {
      "name": "Meta-ExternalAgent",
      "slug": "meta-externalagent",
      "vendor": "Meta",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "Meta-ExternalAgent",
      "userAgent": "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)",
      "tokenOnly": false,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "about": "Collects pages to train Meta’s AI models and to improve its products.",
      "url": "https://canaicrawl.com/bots/meta-externalagent"
    },
    {
      "name": "Bytespider",
      "slug": "bytespider",
      "vendor": "ByteDance",
      "group": "training",
      "purpose": "Model training",
      "robotsToken": "Bytespider",
      "userAgent": "Mozilla/5.0 (Linux; Android 5.0) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; Bytespider; spider-feedback@bytedance.com)",
      "tokenOnly": false,
      "docs": null,
      "about": "ByteDance’s crawler, believed to collect training data for its models. It has no public documentation and is often reported to ignore robots.txt.",
      "url": "https://canaicrawl.com/bots/bytespider"
    },
    {
      "name": "ChatGPT-User",
      "slug": "chatgpt-user",
      "vendor": "OpenAI",
      "group": "fetch",
      "purpose": "Live user request",
      "robotsToken": "ChatGPT-User",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
      "tokenOnly": false,
      "docs": "https://platform.openai.com/docs/bots",
      "about": "Fetches a page when a ChatGPT user asks about it or a GPT action opens it. It does not crawl on its own.",
      "url": "https://canaicrawl.com/bots/chatgpt-user"
    },
    {
      "name": "Claude-User",
      "slug": "claude-user",
      "vendor": "Anthropic",
      "group": "fetch",
      "purpose": "Live user request",
      "robotsToken": "Claude-User",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-User/1.0; +Claude-User@anthropic.com)",
      "tokenOnly": false,
      "docs": "https://support.claude.com/en/articles/8896518",
      "about": "Fetches a page when a Claude user asks about it. It does not crawl on its own.",
      "url": "https://canaicrawl.com/bots/claude-user"
    },
    {
      "name": "Perplexity-User",
      "slug": "perplexity-user",
      "vendor": "Perplexity",
      "group": "fetch",
      "purpose": "Live user request",
      "robotsToken": "Perplexity-User",
      "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)",
      "tokenOnly": false,
      "docs": "https://docs.perplexity.ai/guides/bots",
      "about": "Fetches a page when a Perplexity user asks about it. Perplexity says it generally ignores robots.txt for these requests because a person asked.",
      "url": "https://canaicrawl.com/bots/perplexity-user"
    },
    {
      "name": "MistralAI-User",
      "slug": "mistralai-user",
      "vendor": "Mistral",
      "group": "fetch",
      "purpose": "Live user request",
      "robotsToken": "MistralAI-User",
      "userAgent": "Mozilla/5.0 (compatible; MistralAI-User/1.0; +https://docs.mistral.ai/robots)",
      "tokenOnly": false,
      "docs": "https://docs.mistral.ai/robots/",
      "about": "Fetches a page when a user of Le Chat asks about it. Mistral says it is not used for training.",
      "url": "https://canaicrawl.com/bots/mistralai-user"
    },
    {
      "name": "Meta-ExternalFetcher",
      "slug": "meta-externalfetcher",
      "vendor": "Meta",
      "group": "fetch",
      "purpose": "Live user request",
      "robotsToken": "Meta-ExternalFetcher",
      "userAgent": "meta-externalfetcher/1.1",
      "tokenOnly": false,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "about": "Fetches a page when a user of Meta AI asks about it. Meta says it may bypass robots.txt because a person asked.",
      "url": "https://canaicrawl.com/bots/meta-externalfetcher"
    },
    {
      "name": "AhrefsBot",
      "slug": "ahrefsbot",
      "vendor": "Ahrefs",
      "group": "seo",
      "purpose": "Backlink & SEO index",
      "robotsToken": "AhrefsBot",
      "userAgent": "Mozilla/5.0 (compatible; AhrefsBot/7.0; +http://ahrefs.com/robot/)",
      "tokenOnly": false,
      "docs": "https://ahrefs.com/robot",
      "about": "Builds Ahrefs’ backlink index. Blocking it hides your site from Ahrefs users’ reports, not from search engines.",
      "url": "https://canaicrawl.com/bots/ahrefsbot"
    },
    {
      "name": "SemrushBot",
      "slug": "semrushbot",
      "vendor": "Semrush",
      "group": "seo",
      "purpose": "Backlink & SEO index",
      "robotsToken": "SemrushBot",
      "userAgent": "Mozilla/5.0 (compatible; SemrushBot/7~bl; +http://www.semrush.com/bot.html)",
      "tokenOnly": false,
      "docs": "https://www.semrush.com/bot/",
      "about": "Builds Semrush’s backlink and site audit data. Blocking it does not affect search engines.",
      "url": "https://canaicrawl.com/bots/semrushbot"
    },
    {
      "name": "MJ12bot",
      "slug": "mj12bot",
      "vendor": "Majestic",
      "group": "seo",
      "purpose": "Backlink & SEO index",
      "robotsToken": "MJ12bot",
      "userAgent": "Mozilla/5.0 (compatible; MJ12bot/v1.4.8; http://mj12bot.com/)",
      "tokenOnly": false,
      "docs": "https://mj12bot.com/",
      "about": "Majestic’s distributed crawler for its link index. Blocking it does not affect search engines.",
      "url": "https://canaicrawl.com/bots/mj12bot"
    },
    {
      "name": "DotBot",
      "slug": "dotbot",
      "vendor": "Moz",
      "group": "seo",
      "purpose": "Backlink & SEO index",
      "robotsToken": "DotBot",
      "userAgent": "Mozilla/5.0 (compatible; DotBot/1.2; +https://opensiteexplorer.org/dotbot; help@moz.com)",
      "tokenOnly": false,
      "docs": "https://moz.com/help/moz-procedures/crawlers/dotbot",
      "about": "Moz’s crawler for its link index. Blocking it does not affect search engines.",
      "url": "https://canaicrawl.com/bots/dotbot"
    }
  ]
}
