# Public content may be indexed, cited, and used for live AI answers. # Model-training crawlers are excluded except Google-Extended because # Google uses that product token for both Gemini grounding and training. User-agent: * Content-Signal: search=yes, ai-input=yes, ai-train=no, use=reference Allow: / Disallow: /api/ Disallow: /healthz # Search indexing and user-directed retrieval User-agent: OAI-SearchBot User-agent: ChatGPT-User User-agent: Claude-SearchBot User-agent: Claude-User User-agent: PerplexityBot User-agent: Perplexity-User User-agent: DuckAssistBot User-agent: DuckDuckBot User-agent: MistralAI-User User-agent: Applebot User-agent: Amzn-SearchBot User-agent: Amzn-User User-agent: Slurp User-agent: AhrefsBot User-agent: MojeekBot User-agent: meta-externalfetcher User-agent: Googlebot User-agent: Bingbot User-agent: Google-Agent User-agent: Google-GeminiNotebook User-agent: CloudflareBrowserRenderingCrawler Content-Signal: search=yes, ai-input=yes, ai-train=no, use=reference Allow: / Disallow: /api/ Disallow: /healthz # Gemini grounding and training share the Google-Extended control token. User-agent: Google-Extended Content-Signal: search=yes, ai-input=yes, ai-train=yes, use=reference Allow: / Disallow: /api/ Disallow: /healthz # Training and corpus crawlers User-agent: GPTBot User-agent: ClaudeBot User-agent: CCBot User-agent: Bytespider User-agent: Applebot-Extended User-agent: meta-externalagent User-agent: FacebookBot User-agent: Amazonbot Content-Signal: search=no, ai-input=no, ai-train=no, use=immediate Disallow: / Sitemap: https://marfi.ai/sitemap.xml Sitemap: https://marfi.ai/blog/sitemap.xml