# As a condition of accessing this website, you agree to abide by the following
# content signals:
# (a) If a Content-Signal = yes, you may collect content for the corresponding
# use.
# (b) If a Content-Signal = no, you may not collect content for the
# corresponding use.
# (c) If the website operator does not include a Content-Signal for a
# corresponding use, the website operator neither grants nor restricts
# permission via Content-Signal with respect to the corresponding use.
# The content signals and their meanings are:
# search: building a search index and providing search results (e.g., returning
# hyperlinks and short excerpts from your website's contents). Search does not
# include providing AI-generated search summaries.
# ai-input: inputting content into one or more AI models (e.g., retrieval
# augmented generation, grounding, or other real-time taking of content for
# generative AI search answers).
# ai-train: training or fine-tuning AI models.
# ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS RESERVATIONS OF
# RIGHTS UNDER ARTICLE 4 OF THE EUROPEAN UNION DIRECTIVE 2019/790 ON COPYRIGHT
# AND RELATED RIGHTS IN THE DIGITAL SINGLE MARKET.
# BEGIN Cloudflare Managed content
User-agent: *
Content-Signal: search=yes,ai-train=no
Allow: /
User-agent: Amazonbot
Disallow: /
User-agent: Applebot-Extended
Disallow: /
User-agent: Bytespider
Disallow: /
User-agent: CCBot
Disallow: /
User-agent: ClaudeBot
Disallow: /
User-agent: CloudflareBrowserRenderingCrawler
Disallow: /
User-agent: Google-Extended
Disallow: /
User-agent: GPTBot
Disallow: /
User-agent: meta-externalagent
Disallow: /
# END Cloudflare Managed Content
# robots.txt for imagen-ai.com
# Last updated: 2026-05-13
# Strategy: GEO-friendly — allow AI search/citation crawlers, block aggressive scrapers, retain training opt-out via Content-Signal
#
# Content signal semantics (see C2PA / Cloudflare spec):
# search = building a search index (e.g., Google, Bing)
# ai-input = real-time grounding for AI answers (e.g., AI Overviews, ChatGPT Search, Perplexity)
# ai-train = training or fine-tuning AI models
#
# ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS RESERVATIONS OF
# RIGHTS UNDER ARTICLE 4 OF EU DIRECTIVE 2019/790 ON COPYRIGHT IN THE DIGITAL
# SINGLE MARKET.
# ===========================================================================
# DEFAULT: All crawlers — allow search + AI grounding, opt out of AI training
# ===========================================================================
User-agent: *
Content-Signal: search=yes, ai-input=yes, ai-train=no
Allow: /
# Block WordPress system paths (no SEO value, may leak data)
Disallow: /wp-admin/
Disallow: /wp-includes/
Disallow: /wp-content/plugins/
Disallow: /wp-content/themes/
Disallow: /wp-content/uploads/wpforms/
Disallow: /wp-json/
Disallow: /xmlrpc.php
Disallow: /readme.html
Disallow: /license.txt
# Block admin AJAX (exception)
Allow: /wp-admin/admin-ajax.php
# Block search results, feeds, trackbacks (duplicate/thin content)
Disallow: /?s=
Disallow: /search/
Disallow: /*?replytocom=
Disallow: /trackback/
Disallow: /*/trackback/
Disallow: /comments/feed/
Disallow: /*/feed/
Disallow: /feed/
# Block UTM and tracking parameters from being indexed
Disallow: /*?utm_
Disallow: /*?gclid=
Disallow: /*?fbclid=
Disallow: /*?msclkid=
Disallow: /*?_branch_match_id=
Disallow: /*&utm_
# Block staging / preview / test paths
Disallow: /staging/
Disallow: /preview/
Disallow: /test/
Disallow: /dev/
# ===========================================================================
# MAJOR SEARCH ENGINES — explicit allow
# ===========================================================================
User-agent: Googlebot
Allow: /
Disallow: /wp-admin/
Allow: /wp-admin/admin-ajax.php
User-agent: Googlebot-Image
Allow: /
User-agent: Googlebot-Video
Allow: /
User-agent: Googlebot-News
Allow: /
User-agent: Bingbot
Allow: /
Disallow: /wp-admin/
Allow: /wp-admin/admin-ajax.php
User-agent: DuckDuckBot
Allow: /
User-agent: YandexBot
Allow: /
Crawl-delay: 5
User-agent: Baiduspider
Allow: /
Crawl-delay: 10
User-agent: Applebot
Allow: /
Disallow: /wp-admin/
Allow: /wp-admin/admin-ajax.php
# ===========================================================================
# AI SEARCH & CITATION BOTS — ALLOW (GEO / AIO visibility)
# These bots fetch content to answer user queries and cite sources.
# Allowing them gives Imagen brand mentions in AI Overviews, ChatGPT,
# Perplexity, Claude web search, and Bing Copilot.
# ===========================================================================
# Google AI Overviews / Gemini grounding
User-agent: Google-Extended
Allow: /
# OpenAI — user-initiated browsing (ChatGPT browse with Bing / Atlas)
User-agent: ChatGPT-User
Allow: /
# OpenAI — search index for ChatGPT Search
User-agent: OAI-SearchBot
Allow: /
# Anthropic — Claude user-initiated browsing
User-agent: Claude-User
Allow: /
# Anthropic — Claude web search index
User-agent: Claude-SearchBot
Allow: /
# Perplexity — user-initiated answer engine
User-agent: PerplexityBot
Allow: /
User-agent: Perplexity-User
Allow: /
# You.com
User-agent: YouBot
Allow: /
# Cohere
User-agent: cohere-ai
Allow: /
User-agent: cohere-training-data-crawler
Disallow: /
# Bing Copilot (already covered by Bingbot, but explicit for clarity)
User-agent: BingPreview
Allow: /
# Meta AI search (user-initiated)
User-agent: Meta-ExternalFetcher
Allow: /
# Mistral
User-agent: MistralAI-User
Allow: /
# Diffbot — research / grounding
User-agent: Diffbot
Allow: /
# Common Crawl — used by many open-source AI projects. Allow for visibility
# but opt out of training via Content-Signal at top.
User-agent: CCBot
Allow: /
# ===========================================================================
# AI TRAINING BOTS — DISALLOW
# These bots collect data primarily to train foundation models.
# Disallow to protect proprietary content (Imagen brand IP, blog, talents).
# ===========================================================================
# OpenAI training crawler
User-agent: GPTBot
Disallow: /
# Anthropic training crawler
User-agent: ClaudeBot
Disallow: /
User-agent: anthropic-ai
Disallow: /
# ByteDance training crawler
User-agent: Bytespider
Disallow: /
# Amazon training crawler
User-agent: Amazonbot
Disallow: /
# Apple training crawler (separate from Applebot which is for Spotlight/Siri search)
User-agent: Applebot-Extended
Disallow: /
# Meta training crawler
User-agent: meta-externalagent
Disallow: /
User-agent: FacebookBot
Disallow: /
# Hugging Face training data crawler
User-agent: ImagesiftBot
Disallow: /
# Omgili / Webz training crawler
User-agent: Omgilibot
Disallow: /
User-agent: Omgili
Disallow: /
# AI2 (Allen Institute)
User-agent: AI2Bot
Disallow: /
# Timpi
User-agent: Timpibot
Disallow: /
# Kangaroo
User-agent: PanguBot
Disallow: /
# ===========================================================================
# AGGRESSIVE / LOW-VALUE SCRAPERS — BLOCK ENTIRELY
# ===========================================================================
User-agent: AhrefsBot
Disallow: /
User-agent: SemrushBot
Disallow: /
User-agent: SemrushBot-SA
Disallow: /
User-agent: MJ12bot
Disallow: /
User-agent: DotBot
Disallow: /
User-agent: rogerbot
Disallow: /
User-agent: Screaming Frog SEO Spider
Disallow: /
User-agent: BLEXBot
Disallow: /
User-agent: SeznamBot
Disallow: /
User-agent: PetalBot
Disallow: /
# Bandwidth-heavy / scraping resellers
User-agent: 008
Disallow: /
User-agent: SiteSnagger
Disallow: /
User-agent: WebCopier
Disallow: /
User-agent: WebStripper
Disallow: /
User-agent: HTTrack
Disallow: /
User-agent: Offline Explorer
Disallow: /
User-agent: TurnitinBot
Disallow: /
# ===========================================================================
# SITEMAPS
# ===========================================================================
Sitemap: https://imagen-ai.com/sitemap.xml
Sitemap: https://imagen-ai.com/sitemaps.xml
Sitemap: https://imagen-ai.com/sitemap_index.xml
Sitemap: https://imagen-ai.com/post-sitemap.xml
Sitemap: https://imagen-ai.com/page-sitemap.xml
Sitemap: https://imagen-ai.com/talents-sitemap.xml
Sitemap: https://imagen-ai.com/tools-sitemap.xml
# LLM-specific resources (non-standard but emerging convention)
# llms.txt — concise navigation map for AI
# llms-full.txt — full expanded content for AI grounding
# https://imagen-ai.com/llms.txt
# https://imagen-ai.com/llms-full.txt
# Host directive (Yandex)
Host: https://imagen-ai.com