{"dataset_version":"2026-08-26","generated_at":"2026-09-23T07:29:22Z","methodology_url":"https://citehustle.com/methodology","entries":[{"slug":"gptbot","name":"GPTBot","user_agent_token":"GPTBot","operator":"OpenAI","product_surface":"ChatGPT model training","category":"Training crawler","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"OpenAI crawler documentation","url":"https://developers.openai.com/api/docs/bots"}],"canonical_url":"https://citehustle.com/ai-crawlers/gptbot"},{"slug":"oai-searchbot","name":"OAI-SearchBot","user_agent_token":"OAI-SearchBot","operator":"OpenAI","product_surface":"ChatGPT search citations","category":"Search indexer","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"OpenAI crawler documentation","url":"https://developers.openai.com/api/docs/bots"}],"canonical_url":"https://citehustle.com/ai-crawlers/oai-searchbot"},{"slug":"chatgpt-user","name":"ChatGPT-User","user_agent_token":"ChatGPT-User","operator":"OpenAI","product_surface":"ChatGPT live page fetches","category":"User-initiated fetcher","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"OpenAI crawler documentation","url":"https://developers.openai.com/api/docs/bots"}],"canonical_url":"https://citehustle.com/ai-crawlers/chatgpt-user"},{"slug":"claudebot","name":"ClaudeBot","user_agent_token":"ClaudeBot","operator":"Anthropic","product_surface":"Claude model training","category":"Training crawler","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Anthropic crawler documentation","url":"https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"}],"canonical_url":"https://citehustle.com/ai-crawlers/claudebot"},{"slug":"claude-searchbot","name":"Claude-SearchBot","user_agent_token":"Claude-SearchBot","operator":"Anthropic","product_surface":"Claude search citations","category":"Search indexer","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Anthropic crawler documentation","url":"https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"}],"canonical_url":"https://citehustle.com/ai-crawlers/claude-searchbot"},{"slug":"claude-user","name":"Claude-User","user_agent_token":"Claude-User","operator":"Anthropic","product_surface":"Claude live page fetches","category":"User-initiated fetcher","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Anthropic crawler documentation","url":"https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"}],"canonical_url":"https://citehustle.com/ai-crawlers/claude-user"},{"slug":"googlebot","name":"Googlebot","user_agent_token":"Googlebot","operator":"Google","product_surface":"Google Search + AI Overviews","category":"Search indexer","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Google crawler documentation","url":"https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"}],"canonical_url":"https://citehustle.com/ai-crawlers/googlebot"},{"slug":"google-extended","name":"Google-Extended","user_agent_token":"Google-Extended","operator":"Google","product_surface":"Gemini training & grounding","category":"Training-control token","robots":{"state":"yes","label":"Honors robots.txt","caveat":"It is a robots.txt control token, not a separate crawler. It has no user agent of its own."},"verified_on":"2026-08-26","sources":[{"label":"Google crawler documentation","url":"https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"}],"canonical_url":"https://citehustle.com/ai-crawlers/google-extended"},{"slug":"bingbot","name":"Bingbot","user_agent_token":"bingbot","operator":"Microsoft","product_surface":"Bing Search + Microsoft Copilot","category":"Search indexer","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Bing crawler documentation","url":"https://www.bing.com/webmasters/help/help/which-crawlers-does-bing-use-8c184ec0"}],"canonical_url":"https://citehustle.com/ai-crawlers/bingbot"},{"slug":"perplexitybot","name":"PerplexityBot","user_agent_token":"PerplexityBot","operator":"Perplexity","product_surface":"Perplexity search index","category":"Search indexer","robots":{"state":"yes","label":"Honors robots.txt","caveat":"Perplexity states PerplexityBot honors robots.txt; this was disputed by Cloudflare in 2024 to 2026."},"verified_on":"2026-08-26","sources":[{"label":"Perplexity crawler documentation","url":"https://docs.perplexity.ai/docs/resources/perplexity-crawlers"}],"canonical_url":"https://citehustle.com/ai-crawlers/perplexitybot"},{"slug":"perplexity-user","name":"Perplexity-User","user_agent_token":"Perplexity-User","operator":"Perplexity","product_surface":"Perplexity live answer fetches","category":"User-initiated fetcher","robots":{"state":"partial","label":"Partially / disputed","caveat":"Perplexity has argued user-initiated fetches are not standard crawling; honoring of robots.txt for Perplexity-User has been disputed."},"verified_on":"2026-08-26","sources":[{"label":"Perplexity crawler documentation","url":"https://docs.perplexity.ai/docs/resources/perplexity-crawlers"}],"canonical_url":"https://citehustle.com/ai-crawlers/perplexity-user"},{"slug":"amazonbot","name":"Amazonbot","user_agent_token":"Amazonbot","operator":"Amazon","product_surface":"Alexa & Amazon AI answers","category":"Training crawler","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Amazonbot documentation","url":"https://developer.amazon.com/amazonbot"}],"canonical_url":"https://citehustle.com/ai-crawlers/amazonbot"},{"slug":"applebot-extended","name":"Applebot-Extended","user_agent_token":"Applebot-Extended","operator":"Apple","product_surface":"Apple Intelligence training opt-out","category":"Training-control token","robots":{"state":"yes","label":"Honors robots.txt","caveat":"A robots.txt control token layered on top of Applebot. It has no separate user agent or IP range."},"verified_on":"2026-08-26","sources":[{"label":"Applebot documentation","url":"https://support.apple.com/119829"}],"canonical_url":"https://citehustle.com/ai-crawlers/applebot-extended"},{"slug":"bytespider","name":"Bytespider","user_agent_token":"Bytespider","operator":"ByteDance","product_surface":"ByteDance / Doubao AI training","category":"Training crawler","robots":{"state":"unverified","label":"Unverified","caveat":"Historically reported to crawl aggressively and disregard robots.txt; behavior has been inconsistent, so verify with server logs."},"verified_on":"2026-08-26","sources":[{"label":"IMC 2025 independent crawler study","url":"https://www.sysnet.ucsd.edu/~voelker/pubs/robots-imc25.pdf"}],"canonical_url":"https://citehustle.com/ai-crawlers/bytespider"},{"slug":"ccbot","name":"CCBot","user_agent_token":"CCBot","operator":"Common Crawl","product_surface":"Common Crawl dataset (feeds many LLMs)","category":"Open dataset","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Common Crawl CCBot documentation","url":"https://commoncrawl.org/ccbot"}],"canonical_url":"https://citehustle.com/ai-crawlers/ccbot"},{"slug":"meta-externalagent","name":"Meta-ExternalAgent","user_agent_token":"Meta-ExternalAgent","operator":"Meta","product_surface":"Meta AI / Llama training","category":"Training crawler","robots":{"state":"yes","label":"Honors robots.txt"},"verified_on":"2026-08-26","sources":[{"label":"Meta crawler documentation","url":"https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"}],"canonical_url":"https://citehustle.com/ai-crawlers/meta-externalagent"}]}