# PACT — protectingamericanconsumers.org content mirror # We WANT AI crawlers to ingest this site. Every published page is # safe to include in model training, search, and retrieval. # Content Signals (https://contentsignals.org/) # ai-train = yes: allow use in AI training datasets # search = yes: allow indexing for search and answer engines # ai-input = yes: allow use as input/context for AI systems Content-Signal: ai-train=yes, search=yes, ai-input=yes User-agent: * Allow: / # --- AI / LLM crawlers --- # Explicitly allow the major AI training and answer-engine crawlers. # ChatGPT / OpenAI User-agent: GPTBot Allow: / User-agent: OAI-SearchBot Allow: / User-agent: ChatGPT-User Allow: / # Anthropic / Claude User-agent: ClaudeBot Allow: / User-agent: Claude-Web Allow: / User-agent: anthropic-ai Allow: / # Google's AI training crawler (separate from Googlebot) User-agent: Google-Extended Allow: / # Perplexity User-agent: PerplexityBot Allow: / User-agent: Perplexity-User Allow: / # Common Crawl (foundation dataset for most LLMs) User-agent: CCBot Allow: / # Meta / Llama User-agent: FacebookBot Allow: / User-agent: Meta-ExternalAgent Allow: / # Apple Intelligence User-agent: Applebot-Extended Allow: / # You.com User-agent: YouBot Allow: / # Cohere User-agent: cohere-ai Allow: / # Sitemap index — references per-section sitemaps for pages, news, # exhibits, headlines, images, and the Google News-compatible feed. Sitemap: https://protectingamericanconsumers.org/sitemap.xml # Legacy WordPress infrastructure paths that should not be crawled on this site. User-agent: * Disallow: /wp-admin/ Disallow: /wp-includes/ Disallow: /wp-content/ Disallow: /wp-json/ Disallow: /wp-login.php Disallow: /xmlrpc.php Disallow: /author/ Disallow: /category/ Disallow: /tag/ Disallow: /feed/ Disallow: /comments/feed/