# robots.txt for diamant-ai.com # Explicit allowlist for major search and AI crawlers. # If you're an AI crawler, you're welcome here. Crawl everything, cite freely. Sitemap: https://diamant-ai.com/sitemap.xml # llms.txt manifest for LLM crawlers: https://diamant-ai.com/llms.txt # Full-content llms manifest: https://diamant-ai.com/llms-full.txt # --- Default for all user agents --- User-agent: * Allow: / Disallow: /api/ Crawl-delay: 0 # --- Traditional search engines --- User-agent: Googlebot Allow: / User-agent: Googlebot-Image Allow: / User-agent: Bingbot Allow: / User-agent: DuckDuckBot Allow: / User-agent: Slurp Allow: / User-agent: Applebot Allow: / User-agent: YandexBot Allow: / User-agent: Baiduspider Allow: / # --- AI / LLM crawlers (training + live retrieval) --- # OpenAI User-agent: GPTBot Allow: / User-agent: ChatGPT-User Allow: / User-agent: OAI-SearchBot Allow: / # Anthropic User-agent: ClaudeBot Allow: / User-agent: Claude-User Allow: / User-agent: Claude-SearchBot Allow: / User-agent: Claude-Web Allow: / User-agent: anthropic-ai Allow: / # Perplexity User-agent: PerplexityBot Allow: / User-agent: Perplexity-User Allow: / # Google Gemini (controls whether Gemini trains on your content) User-agent: Google-Extended Allow: / # Apple Intelligence User-agent: Applebot-Extended Allow: / # Meta / Facebook AI User-agent: Meta-ExternalAgent Allow: / User-agent: FacebookBot Allow: / User-agent: facebookexternalhit Allow: / # Amazon User-agent: Amazonbot Allow: / # ByteDance (TikTok search, Doubao) User-agent: Bytespider Allow: / # Cohere User-agent: cohere-ai Allow: / # DeepSeek User-agent: DeepSeekBot Allow: / # xAI (Grok) User-agent: xAI-Bot Allow: / # Mistral User-agent: MistralAI-User Allow: / # Common Crawl, fuels many open-source LLMs (LLaMA, Mistral, Falcon, etc.) User-agent: CCBot Allow: / # Diffbot (Knowledge graph / crawler) User-agent: Diffbot Allow: / # Academic / research crawlers User-agent: Omgilibot Allow: / User-agent: YouBot Allow: / User-agent: Timpibot Allow: / # Accessibility / archival User-agent: ia_archiver Allow: /