# Optimizing for SearchGPT and blocking GPT bots from training on our data while indexing our content in their search results. # Allowed crawlers: Googlebot, Bingbot, OAI-Searchbot, PerplexityBot, AdsBot-Google, Googlebot-Image, FirecrawlAgent, AhrefsBot, AhrefsSiteAudit. # Pagination URLs (_page=) are intentionally NOT blocked here so crawlers can follow internal links and pass equity to deep content. # Indexation of paginated URLs is controlled via on those pages. User-agent: Googlebot Disallow: /*?r=0 Disallow: /_next/static/ User-agent: Bingbot Disallow: /_next/static/ User-agent: OAI-Searchbot Disallow: /_next/static/ User-agent: PerplexityBot Disallow: /_next/static/ User-agent: AdsBot-Google Allow: / User-agent: Googlebot-Image Allow: / User-agent: FirecrawlAgent Allow: / Disallow: /_next/static/ User-agent: AhrefsSiteAudit Allow: / Disallow: /_next/static/ User-agent: AhrefsBot Allow: / Disallow: /_next/static/ # Block all access for specific AI training bots User-agent: Amazonbot Disallow: / User-agent: Anthropic-ai Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: AwarioRssBot Disallow: / User-agent: AwarioSmartBot Disallow: / User-agent: Bytespider Disallow: / User-agent: CCBot Disallow: / User-agent: ChatGPT-User Disallow: User-agent: ClaudeBot Disallow: / User-agent: Claude-Web Disallow: / User-agent: Cohere-ai Disallow: / User-agent: DataForSeoBot Disallow: / User-agent: FacebookBot Disallow: / User-agent: Google-Extended Disallow: / User-agent: GPTBot Disallow: / User-agent: ImagesiftBot Disallow: / User-agent: Magpie-crawler Disallow: / User-agent: Omgili Disallow: / User-agent: Omgilibot Disallow: / User-agent: Peer39_crawler Disallow: / User-agent: Peer39_crawler/1.0 Disallow: / User-agent: YouBot Disallow: / # Default rules for any other crawler — block tracking params and build artifacts User-agent: * Disallow: /*?via= Disallow: /*?page= Disallow: /*?next= Disallow: /_next/static/ Sitemap: https://www.taxgpt.com/sitemap.xml Sitemap: https://www.taxgpt.com/sitemap.xml