{
  "crawlers": [
    {
      "id": "gptbot",
      "token": "gptbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
      "label": "GPTBot",
      "operator": "OpenAI",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://platform.openai.com/docs/bots",
      "note": "Crawls content that may be used to train OpenAI's generative AI foundation models.",
      "sites_blocking": 493,
      "sites_allowing": 3465,
      "block_rate": 12.5
    },
    {
      "id": "oai-searchbot",
      "token": "oai-searchbot",
      "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
      "label": "OAI-SearchBot",
      "operator": "OpenAI",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://platform.openai.com/docs/bots",
      "note": "Indexes pages so they can be surfaced and cited in ChatGPT search results, not for training.",
      "sites_blocking": 161,
      "sites_allowing": 3797,
      "block_rate": 4.1
    },
    {
      "id": "chatgpt-user",
      "token": "chatgpt-user",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
      "label": "ChatGPT-User",
      "operator": "OpenAI",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://platform.openai.com/docs/bots",
      "note": "Fetches a page when a ChatGPT user or GPT Action asks for it; user-initiated, so robots rules may not apply.",
      "sites_blocking": 235,
      "sites_allowing": 3723,
      "block_rate": 5.9
    },
    {
      "id": "oai-adsbot",
      "token": "oai-adsbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot",
      "label": "OAI-AdsBot",
      "operator": "OpenAI",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://platform.openai.com/docs/bots",
      "note": "Visits pages submitted as ChatGPT ads to check policy compliance and ad relevance; not used for model training.",
      "sites_blocking": 1,
      "sites_allowing": 3957,
      "block_rate": 0
    },
    {
      "id": "claudebot",
      "token": "claudebot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
      "label": "ClaudeBot",
      "operator": "Anthropic",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "note": "Collects web content that may contribute to training Anthropic's models; honors Crawl-delay.",
      "sites_blocking": 434,
      "sites_allowing": 3524,
      "block_rate": 11
    },
    {
      "id": "claude-user",
      "token": "claude-user",
      "ua": "Mozilla/5.0 (compatible; Claude-User/1.0; +Claude-User@anthropic.com)",
      "label": "Claude-User",
      "operator": "Anthropic",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "note": "Retrieves pages on demand when a Claude user's question needs live web content.",
      "sites_blocking": 141,
      "sites_allowing": 3817,
      "block_rate": 3.6
    },
    {
      "id": "claude-searchbot",
      "token": "claude-searchbot",
      "ua": "Mozilla/5.0 (compatible; Claude-SearchBot/1.0; +Claude-SearchBot@anthropic.com)",
      "label": "Claude-SearchBot",
      "operator": "Anthropic",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "note": "Indexes content to improve the relevance and accuracy of Claude's search results.",
      "sites_blocking": 137,
      "sites_allowing": 3821,
      "block_rate": 3.5
    },
    {
      "id": "anthropic-ai",
      "token": "anthropic-ai",
      "ua": "anthropic-ai",
      "label": "anthropic-ai",
      "operator": "Anthropic",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "note": "Legacy token widely blocked for Anthropic training; Anthropic now documents ClaudeBot, Claude-User and Claude-SearchBot.",
      "sites_blocking": 260,
      "sites_allowing": 3698,
      "block_rate": 6.6
    },
    {
      "id": "google-extended",
      "token": "google-extended",
      "ua": "",
      "label": "Google-Extended",
      "operator": "Google",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "note": "Control token with no user agent of its own; governs Gemini training and grounding use of Googlebot data.",
      "sites_blocking": 393,
      "sites_allowing": 3565,
      "block_rate": 9.9
    },
    {
      "id": "googlebot",
      "token": "googlebot",
      "ua": "Mozilla/5.0 (Linux; Android 6.0.1; Nexus 5X Build/MMB29P) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Mobile Safari/537.36 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
      "label": "Googlebot",
      "operator": "Google",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://developers.google.com/search/docs/crawling-indexing/googlebot",
      "note": "Crawls and renders pages for Google Search, Images, Video, News and Discover.",
      "sites_blocking": 4,
      "sites_allowing": 3954,
      "block_rate": 0.1
    },
    {
      "id": "googlebot-news",
      "token": "googlebot-news",
      "ua": "Mozilla/5.0 (Linux; Android 6.0.1; Nexus 5X Build/MMB29P) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/W.X.Y.Z Mobile Safari/537.36 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
      "label": "Googlebot-News",
      "operator": "Google",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "note": "Robots token controlling Google News inclusion; crawling itself uses the Googlebot user agents.",
      "sites_blocking": 9,
      "sites_allowing": 3949,
      "block_rate": 0.2
    },
    {
      "id": "google-cloudvertexbot",
      "token": "google-cloudvertexbot",
      "ua": "Mozilla/5.0 (compatible; Google-CloudVertexBot; +https://cloud.google.com/vertex-ai-bot)",
      "label": "Google-CloudVertexBot",
      "operator": "Google",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "note": "Crawls sites at a site owner's request to build Vertex AI agents; no effect on Google Search.",
      "sites_blocking": 114,
      "sites_allowing": 3844,
      "block_rate": 2.9
    },
    {
      "id": "googleother",
      "token": "googleother",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GoogleOther) Chrome/W.X.Y.Z Safari/537.36",
      "label": "GoogleOther",
      "operator": "Google",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "note": "Generic Google crawler used by product teams for one-off fetches such as internal research and development.",
      "sites_blocking": 79,
      "sites_allowing": 3879,
      "block_rate": 2
    },
    {
      "id": "applebot",
      "token": "applebot",
      "ua": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
      "label": "Applebot",
      "operator": "Apple",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://support.apple.com/en-us/119829",
      "note": "Crawls for Siri, Spotlight and Safari search; falls back to Googlebot rules and ignores Crawl-delay.",
      "sites_blocking": 62,
      "sites_allowing": 3896,
      "block_rate": 1.6
    },
    {
      "id": "applebot-extended",
      "token": "applebot-extended",
      "ua": "",
      "label": "Applebot-Extended",
      "operator": "Apple",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://support.apple.com/en-us/119829",
      "note": "Control token with no user agent; disallowing it excludes crawled content from Apple foundation model training.",
      "sites_blocking": 355,
      "sites_allowing": 3603,
      "block_rate": 9
    },
    {
      "id": "bingbot",
      "token": "bingbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) Chrome/116.0.1938.76 Safari/537.36",
      "label": "Bingbot",
      "operator": "Microsoft",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "note": "Indexes pages for Bing search and the Copilot answers that are grounded in the Bing index.",
      "sites_blocking": 6,
      "sites_allowing": 3952,
      "block_rate": 0.2
    },
    {
      "id": "msnbot",
      "token": "msnbot",
      "ua": "msnbot/1.1 (+http://search.msn.com/msnbot.htm)",
      "label": "msnbot",
      "operator": "Microsoft",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "note": "Legacy Microsoft search crawler token still honored alongside bingbot.",
      "sites_blocking": 4,
      "sites_allowing": 3954,
      "block_rate": 0.1
    },
    {
      "id": "perplexitybot",
      "token": "perplexitybot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
      "label": "PerplexityBot",
      "operator": "Perplexity",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "note": "Indexes and links pages in Perplexity search results; not used to collect foundation model training data.",
      "sites_blocking": 262,
      "sites_allowing": 3696,
      "block_rate": 6.6
    },
    {
      "id": "perplexity-user",
      "token": "perplexity-user",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)",
      "label": "Perplexity-User",
      "operator": "Perplexity",
      "purpose": "assist",
      "respectsRobots": false,
      "docs": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "note": "Fetches a page for a specific user question; Perplexity documents that it generally ignores robots.txt.",
      "sites_blocking": 135,
      "sites_allowing": 3823,
      "block_rate": 3.4
    },
    {
      "id": "meta-externalagent",
      "token": "meta-externalagent",
      "ua": "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/web-crawlers)",
      "label": "Meta-ExternalAgent",
      "operator": "Meta",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "note": "Crawls the web to train Meta's foundation AI models and to index content directly into products.",
      "sites_blocking": 387,
      "sites_allowing": 3571,
      "block_rate": 9.8
    },
    {
      "id": "meta-externalfetcher",
      "token": "meta-externalfetcher",
      "ua": "meta-externalfetcher/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/web-crawlers)",
      "label": "Meta-ExternalFetcher",
      "operator": "Meta",
      "purpose": "assist",
      "respectsRobots": false,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "note": "Fetches individual links for agentic AI tasks; Meta documents that it may bypass robots.txt.",
      "sites_blocking": 154,
      "sites_allowing": 3804,
      "block_rate": 3.9
    },
    {
      "id": "facebookbot",
      "token": "facebookbot",
      "ua": "Mozilla/5.0 (compatible; FacebookBot/1.0; +https://developers.facebook.com/docs/sharing/webmasters/crawler)",
      "label": "FacebookBot",
      "operator": "Meta",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developers.facebook.com/docs/sharing/bot/",
      "note": "Crawls public pages to improve language models behind Meta's speech recognition technology.",
      "sites_blocking": 222,
      "sites_allowing": 3736,
      "block_rate": 5.6
    },
    {
      "id": "meta-webindexer",
      "token": "meta-webindexer",
      "ua": "meta-webindexer/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/web-crawlers)",
      "label": "Meta-WebIndexer",
      "operator": "Meta",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "note": "Indexes pages so Meta AI can cite and link them in its search answers.",
      "sites_blocking": 81,
      "sites_allowing": 3877,
      "block_rate": 2
    },
    {
      "id": "meta-externalads",
      "token": "meta-externalads",
      "ua": "meta-externalads/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/web-crawlers)",
      "label": "Meta-ExternalAds",
      "operator": "Meta",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "note": "Crawls the web to improve Meta's advertising and other business products and services.",
      "sites_blocking": 1,
      "sites_allowing": 3957,
      "block_rate": 0
    },
    {
      "id": "facebookexternalhit",
      "token": "facebookexternalhit",
      "ua": "facebookexternalhit/1.1 (+http://www.facebook.com/externalhit_uatext.php)",
      "label": "facebookexternalhit",
      "operator": "Meta",
      "purpose": "assist",
      "respectsRobots": false,
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "note": "Fetches shared links for Facebook, Instagram and Messenger previews; may bypass robots.txt for integrity checks.",
      "sites_blocking": 14,
      "sites_allowing": 3944,
      "block_rate": 0.4
    },
    {
      "id": "bytespider",
      "token": "bytespider",
      "ua": "Mozilla/5.0 (Linux; Android 5.0) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; Bytespider; spider-feedback@bytedance.com)",
      "label": "Bytespider",
      "operator": "ByteDance",
      "purpose": "train",
      "respectsRobots": false,
      "docs": "https://knownagents.com/agents/bytespider",
      "note": "Downloads content to train ByteDance LLMs and is widely reported to ignore robots.txt directives.",
      "sites_blocking": 470,
      "sites_allowing": 3488,
      "block_rate": 11.9
    },
    {
      "id": "tiktokspider",
      "token": "tiktokspider",
      "ua": "Mozilla/5.0 (Linux; Android 5.0) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; TikTokSpider; ttspider-feedback@tiktok.com)",
      "label": "TikTokSpider",
      "operator": "ByteDance",
      "purpose": "assist",
      "respectsRobots": false,
      "docs": "https://knownagents.com/agents/tiktokspider",
      "note": "Fetches shared URLs for TikTok link previews and feeds; not expected to follow robots.txt.",
      "sites_blocking": 44,
      "sites_allowing": 3914,
      "block_rate": 1.1
    },
    {
      "id": "amazonbot",
      "token": "amazonbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amazonbot/0.1) Chrome/W.X.Y.Z Safari/537.36",
      "label": "Amazonbot",
      "operator": "Amazon",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://developer.amazon.com/amazonbot",
      "note": "Crawls for Amazon product and Alexa answers and may use the content to train Amazon AI models.",
      "sites_blocking": 334,
      "sites_allowing": 3624,
      "block_rate": 8.4
    },
    {
      "id": "amzn-searchbot",
      "token": "amzn-searchbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amzn-SearchBot/0.1) Chrome/W.X.Y.Z Safari/537.36",
      "label": "Amzn-SearchBot",
      "operator": "Amazon",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://developer.amazon.com/amazonbot",
      "note": "Indexes content for Amazon search experiences such as Alexa; does not crawl for generative AI training.",
      "sites_blocking": 36,
      "sites_allowing": 3922,
      "block_rate": 0.9
    },
    {
      "id": "amzn-user",
      "token": "amzn-user",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amzn-User/0.1) Chrome/W.X.Y.Z Safari/537.36",
      "label": "Amzn-User",
      "operator": "Amazon",
      "purpose": "assist",
      "respectsRobots": false,
      "docs": "https://developer.amazon.com/amazonbot",
      "note": "Fetches live pages to answer a user's Alexa question; Amazon documents it may not follow all robots.txt rules.",
      "sites_blocking": 27,
      "sites_allowing": 3931,
      "block_rate": 0.7
    },
    {
      "id": "ccbot",
      "token": "ccbot",
      "ua": "CCBot/2.0 (https://commoncrawl.org/faq/)",
      "label": "CCBot",
      "operator": "Common Crawl Foundation",
      "purpose": "archive",
      "respectsRobots": true,
      "docs": "https://commoncrawl.org/ccbot",
      "note": "Builds the open Common Crawl web archive, a common source of LLM pretraining corpora.",
      "sites_blocking": 509,
      "sites_allowing": 3449,
      "block_rate": 12.9
    },
    {
      "id": "diffbot",
      "token": "diffbot",
      "ua": "Mozilla/5.0 (compatible; Diffbot/1.0; +https://diffbot.com)",
      "label": "Diffbot",
      "operator": "Diffbot",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://docs.diffbot.com/docs/en/guides-diffbot-crawlbot",
      "note": "Extracts structured page data for Diffbot's knowledge graph, which is licensed to AI customers.",
      "sites_blocking": 281,
      "sites_allowing": 3677,
      "block_rate": 7.1
    },
    {
      "id": "omgili",
      "token": "omgili",
      "ua": "omgili/0.5 +http://omgili.com",
      "label": "omgili",
      "operator": "Webz.io",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/",
      "note": "Collects forum, news and blog content that Webz.io sells as web data feeds, including for AI training.",
      "sites_blocking": 268,
      "sites_allowing": 3690,
      "block_rate": 6.8
    },
    {
      "id": "omgilibot",
      "token": "omgilibot",
      "ua": "omgilibot/0.4 +http://omgili.org/crawler",
      "label": "omgilibot",
      "operator": "Webz.io",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/",
      "note": "Legacy Omgili search crawler token still blocked alongside the current omgili agent.",
      "sites_blocking": 287,
      "sites_allowing": 3671,
      "block_rate": 7.3
    },
    {
      "id": "ai2bot",
      "token": "ai2bot",
      "ua": "AI2Bot/1.0",
      "label": "AI2Bot",
      "operator": "Allen Institute for AI",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://allenai.org/crawler",
      "note": "Collects web text for Ai2's open datasets used to train open language models such as OLMo.",
      "sites_blocking": 159,
      "sites_allowing": 3799,
      "block_rate": 4
    },
    {
      "id": "cohere-ai",
      "token": "cohere-ai",
      "ua": "Mozilla/5.0 (compatible; cohere-ai/1.0; +https://cohere.com)",
      "label": "cohere-ai",
      "operator": "Cohere",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://knownagents.com/agents/cohere-ai",
      "note": "Retrieves pages to answer user-initiated prompts in Cohere's enterprise AI products.",
      "sites_blocking": 273,
      "sites_allowing": 3685,
      "block_rate": 6.9
    },
    {
      "id": "cohere-training-data-crawler",
      "token": "cohere-training-data-crawler",
      "ua": "cohere-training-data-crawler",
      "label": "cohere-training-data-crawler",
      "operator": "Cohere",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://knownagents.com/agents/cohere-training-data-crawler",
      "note": "Downloads training data for the large language models behind Cohere's enterprise AI products.",
      "sites_blocking": 152,
      "sites_allowing": 3806,
      "block_rate": 3.8
    },
    {
      "id": "mistralai-user",
      "token": "mistralai-user",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-User/1.0; +https://docs.mistral.ai/robots)",
      "label": "MistralAI-User",
      "operator": "Mistral AI",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://docs.mistral.ai/robots/",
      "note": "Fetches pages on demand so Mistral's Vibe assistant can answer a question with live, cited web content.",
      "sites_blocking": 120,
      "sites_allowing": 3838,
      "block_rate": 3
    },
    {
      "id": "mistralai-index",
      "token": "mistralai-index",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Index/1.0; +https://docs.mistral.ai/robots)",
      "label": "MistralAI-Index",
      "operator": "Mistral AI",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://docs.mistral.ai/robots/",
      "note": "Indexes content for Mistral search behind Vibe answers; not used for generative AI training.",
      "sites_blocking": 16,
      "sites_allowing": 3942,
      "block_rate": 0.4
    },
    {
      "id": "mistralai-training",
      "token": "mistralai-training",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Training/1.0; +https://docs.mistral.ai/robots)",
      "label": "MistralAI-Training",
      "operator": "Mistral AI",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://docs.mistral.ai/robots/",
      "note": "Crawls web content to build datasets for training Mistral's generative AI models.",
      "sites_blocking": 9,
      "sites_allowing": 3949,
      "block_rate": 0.2
    },
    {
      "id": "duckassistbot",
      "token": "duckassistbot",
      "ua": "DuckAssistBot/1.2; (+http://duckduckgo.com/duckassistbot.html)",
      "label": "DuckAssistBot",
      "operator": "DuckDuckGo",
      "purpose": "assist",
      "respectsRobots": true,
      "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
      "note": "Crawls pages in real time for DuckDuckGo's cited AI-assisted answers; not used for model training.",
      "sites_blocking": 150,
      "sites_allowing": 3808,
      "block_rate": 3.8
    },
    {
      "id": "youbot",
      "token": "youbot",
      "ua": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; YouBot/1.0; +https://docs.you.com/youbot) Chrome/142.0.0.0 Safari/537.36",
      "label": "YouBot",
      "operator": "You.com",
      "purpose": "search",
      "respectsRobots": true,
      "docs": "https://docs.you.com/youbot",
      "note": "Indexes pages for You.com search results and the AI answers built on that index.",
      "sites_blocking": 200,
      "sites_allowing": 3758,
      "block_rate": 5.1
    },
    {
      "id": "pangubot",
      "token": "pangubot",
      "ua": "Mozilla/5.0 (Linux; Android 7.0;) AppleWebKit/537.36 (KHTML, like Gecko) Mobile Safari/537.36 (compatible; PanguBot;pangubot@huawei.com)",
      "label": "PanguBot",
      "operator": "Huawei",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://knownagents.com/agents/pangubot",
      "note": "Collects web content used to train Huawei's PanGu family of large models.",
      "sites_blocking": 138,
      "sites_allowing": 3820,
      "block_rate": 3.5
    },
    {
      "id": "timpibot",
      "token": "timpibot",
      "ua": "Timpibot/1.0 (+http://timpi.io/crawler)",
      "label": "Timpibot",
      "operator": "Timpi",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://timpi.io/",
      "note": "Crawls pages for Timpi's decentralized index, which is also used as LLM training data.",
      "sites_blocking": 206,
      "sites_allowing": 3752,
      "block_rate": 5.2
    },
    {
      "id": "imagesiftbot",
      "token": "imagesiftbot",
      "ua": "Mozilla/5.0 (compatible; ImagesiftBot; +imagesift.com)",
      "label": "ImagesiftBot",
      "operator": "ImageSift (Hive)",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://imagesift.com/about",
      "note": "Downloads public images plus surrounding text to build ImageSift's searchable image index.",
      "sites_blocking": 180,
      "sites_allowing": 3778,
      "block_rate": 4.5
    },
    {
      "id": "kangaroo-bot",
      "token": "kangaroo bot",
      "ua": "Kangaroo Bot",
      "label": "Kangaroo Bot",
      "operator": "Kangaroo LLM",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://knownagents.com/agents/kangaroo-bot",
      "note": "Scrapes site content into datasets used to train the Kangaroo LLM.",
      "sites_blocking": 123,
      "sites_allowing": 3835,
      "block_rate": 3.1
    },
    {
      "id": "semrushbot-ocob",
      "token": "semrushbot-ocob",
      "ua": "Mozilla/5.0 (compatible; SemrushBot-OCOB/1; +https://www.semrush.com/bot/)",
      "label": "SemrushBot-OCOB",
      "operator": "Semrush",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://www.semrush.com/bot/",
      "note": "Crawls pages to feed Semrush's ContentShake AI writing tool.",
      "sites_blocking": 136,
      "sites_allowing": 3822,
      "block_rate": 3.4
    },
    {
      "id": "scrapy",
      "token": "scrapy",
      "ua": "Scrapy/2.17.0 (+https://scrapy.org)",
      "label": "Scrapy",
      "operator": "Zyte (open-source framework)",
      "purpose": "train",
      "respectsRobots": true,
      "docs": "https://scrapy.org/",
      "note": "Generic scraping framework often used to build AI training datasets; obeys robots.txt only when ROBOTSTXT_OBEY is on.",
      "sites_blocking": 175,
      "sites_allowing": 3783,
      "block_rate": 4.4
    }
  ]
}