# AI training crawlers blocked. Citation / on-demand / search crawlers
# (ChatGPT-User, OAI-SearchBot, Claude-User, Claude-SearchBot, PerplexityBot,
# DuckAssistBot, MistralAI-User, AI agents like Operator/Devin/NovaAct/Mariner,
# Googlebot, Bingbot, …) are intentionally NOT listed. They bring traffic via
# AI-Overviews, ChatGPT-Search, Claude-Search, Perplexity-Citations etc.
# Bot list aligned with corporate policy; non-training crawlers (SEO tools,
# web archives) are out of scope for this policy.
User-agent: AI2Bot
Disallow: /
User-agent: AllenAI
Disallow: /
User-agent: anthropic-ai
Disallow: /
User-agent: Applebot-Extended
Disallow: /
User-agent: BedrockBot
Disallow: /
User-agent: Bytespider
Disallow: /
User-agent: CCBot
Disallow: /
User-agent: ChatGLM-Spider
Disallow: /
User-agent: ClaudeBot
Disallow: /
User-agent: CloudVertexBot
Disallow: /
User-agent: cohere-ai
Disallow: /
User-agent: Cohere-Training-Data-Crawler
Disallow: /
User-agent: Cotoyogi
Disallow: /
User-agent: DeepSeek
Disallow: /
User-agent: DeepSeekBot
Disallow: /
User-agent: FacebookBot
Disallow: /
User-agent: Google-Extended
Disallow: /
User-agent: GPTBot
Disallow: /
User-agent: ICC-Crawler
Disallow: /
User-agent: ImagesiftBot
Disallow: /
User-agent: img2dataset
Disallow: /
User-agent: laion-huggingface-processor
Disallow: /
User-agent: LCC
Disallow: /
User-agent: meta-externalagent
Disallow: /
User-agent: meta-externalfetcher
Disallow: /
User-agent: omgili
Disallow: /
User-agent: omgilibot
Disallow: /
User-agent: PanguBot
Disallow: /
User-agent: PetalBot
Disallow: /
User-agent: PhindBot
Disallow: /
User-agent: Poseidon Research Crawler
Disallow: /
User-agent: SBIntuitionsBot
Disallow: /
User-agent: TikTokSpider
Disallow: /
User-agent: Timpibot
Disallow: /
User-agent: Webzio-Extended
Disallow: /
User-agent: YandexAdditional
Disallow: /
User-agent: YandexAdditionalBot
Disallow: /
User-agent: YouBot
Disallow: /
# Catch-all for everyone else (Googlebot, Bingbot, ChatGPT-User, OAI-SearchBot,
# Claude-User, Claude-SearchBot, PerplexityBot, DuckAssistBot, MistralAI-User,
# AI agents, …). The Content-Signal makes the policy explicit for crawlers that
# follow Cloudflare's Content Signals spec: search results yes, AI citations
# yes, AI training no.
User-agent: *
Content-Signal: search=yes, ai-input=yes, ai-train=no
Disallow: