User-agent: *
Content-Signal: search=yes, ai-input=yes, ai-train=no
Allow: /
# Content Signals interpretation:
# - search=yes: search indexing and short excerpts are allowed
# - ai-input=yes: real-time retrieval and grounding for agent answers are allowed
# - ai-train=no: model training and fine-tuning are not allowed
# ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS RESERVATIONS OF
# RIGHTS UNDER ARTICLE 4 OF THE EUROPEAN UNION DIRECTIVE 2019/790.
# ── AI retrieval agents - allowed ─────────────────────────────────────────────
# These bots power real-time answers and search, not training datasets.
# ChatGPT browsing (retrieval, not training)
User-agent: ChatGPT-User
Allow: /
# Anthropic retrieval
User-agent: Anthropic-AI
Allow: /
# Anthropic web crawler
User-agent: ClaudeBot
Allow: /
# Perplexity search
User-agent: PerplexityBot
Allow: /
# Amazon Alexa / Rufus retrieval
User-agent: Amazonbot
Allow: /
# ── AI training crawlers - disallowed ─────────────────────────────────────────
# These bots harvest content for training datasets.
# Disallow is the enforceable opt-out; X-Robots-Tag noai is an additional signal.
# OpenAI training crawler
User-agent: GPTBot
Disallow: /
# Common Crawl - used as training data by many LLMs
User-agent: CCBot
Disallow: /
# Google Gemini / AI Overviews training opt-out
User-agent: Google-Extended
Disallow: /
# ByteDance / TikTok training crawler
User-agent: Bytespider
Disallow: /
# Image dataset harvesting
User-agent: ImagesiftBot
Disallow: /
Sitemap: https://guitarwiz.app/sitemap-index.xml
Sitemap: https://guitarwiz.app/chords-sitemap.xml