User-agent: * Disallow: /clip/* Disallow: /auth* Disallow: /shows* Disallow: /login Disallow: /all-tv-shows Disallow: /profile #----------------------------- # OpenAI/ChatGPT --------- # Allowing OpenAI ChatGPT users and search queries NOT used for training LLMs User-agent: ChatGPT-User Allow: / User-agent: OAI-SearchBot Allow: / # Disallowing the OpenAI ChatGPT web crawler used to train LLMs User-agent: GPTBot Disallow: / #----------------------------- #Claude/anthropic ------------ # Allowing anthropic/Claude users and search queries NOT used for training LLMs User-agent: Claude-User Allow: / User-agent: Claude-SearchBot Allow: / # Disallowing the anthropic/Claude web crawler used to train LLMs User-agent: ClaudeBot Disallow: / #----------------------------- # Disallowing the Perplexity web crawler used for search indexing User-agent: PerplexityBot Allow: / # Allowing Perplexity users and search queries NOT used for training LLMs User-agent: PerplexityUser Allow: / #----------------------------- # Disallowing Common Crawl User-agent: CCBot Disallow: / # Disallowing Google Bard and Vertex AI web crawlers User-agent: Google-Extended Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: Bytespider Disallow: / User-agent: cohere-ai Disallow: / User-agent: Diffbot Disallow: / User-agent: omgili Disallow: / User-agent: omgilibot Disallow: / # Block Meta's dedicated AI training crawlers User-agent: Meta-ExternalAgent Disallow: / User-agent: Meta-ExternalFetcher Disallow: / # Block the generic "FacebookBot" which Meta uses for training User-agent: FacebookBot Disallow: / # Explicitly ALLOW the bot that creates link previews User-agent: facebookexternalhit Allow: / Sitemap: https://www.mainepublic.org/sitemap.xml Sitemap: https://www.mainepublic.org/sitemap-latest.xml Sitemap: https://www.mainepublic.org/news-sitemap-content.xml