# ============================================================ # robots.txt for sarkariresult.com.ht # Final Version — June 2026 # Deep audited against official documentation: # Google (April 2026), Anthropic (Feb 2026), OpenAI (Dec 2025), # Amazon (June 2026), Apple, DuckDuckGo, Perplexity, Meta # Cloudflare robots.txt cross-checked and conflicts resolved # Goal: Maximum indexing + AI citations # Block ONLY verified AI training bots # No Crawl-Delay — time-critical Sarkari Results content # ============================================================ # ============================================================ # SECTION 1: GOOGLE — ALL OFFICIAL CRAWLERS # Source: developers.google.com/crawling (April 2026) # ============================================================ User-agent: Googlebot Allow: / User-agent: Googlebot-Image Allow: / User-agent: Googlebot-Video Allow: / User-agent: Googlebot-News Allow: / User-agent: Googlebot-Mobile Allow: / User-agent: Google-InspectionTool Allow: / User-agent: Storebot-Google Allow: / User-agent: GoogleOther Allow: / User-agent: GoogleOther-Image Allow: / User-agent: GoogleOther-Video Allow: / User-agent: Google-CloudVertexBot Allow: / User-agent: AdsBot-Google Allow: / User-agent: Mediapartners-Google Allow: / # Google-Extended: Controls AI Overviews, Gemini, AI Mode, # Vertex AI grounding, NotebookLM. # ALLOW = content eligible for AI Overview citations. # Cloudflare default BLOCKS this — we OVERRIDE to Allow # because AI Overview visibility = more traffic for your site. User-agent: Google-Extended Allow: / # ============================================================ # SECTION 2: BING / MICROSOFT (Search + Copilot AI) # ============================================================ User-agent: Bingbot Allow: / User-agent: msnbot Allow: / User-agent: msnbot-media Allow: / User-agent: BingPreview Allow: / User-agent: MSNBot Allow: / # ============================================================ # SECTION 3: YAHOO # ============================================================ User-agent: Slurp Allow: / # ============================================================ # SECTION 4: YANDEX (Search + Alice AI assistant) # ============================================================ User-agent: YandexBot Allow: / User-agent: YandexImages Allow: / User-agent: YandexVideo Allow: / User-agent: YandexNews Allow: / User-agent: YandexMetrika Allow: / User-agent: YandexWebmaster Allow: / User-agent: YandexGPTBot Allow: / # ============================================================ # SECTION 5: DUCKDUCKGO (Search + DuckAssist AI) # ============================================================ User-agent: DuckDuckBot Allow: / # DuckAssistBot: DuckDuckGo AI answers — retrieval ONLY, NOT training User-agent: DuckAssistBot Allow: / # ============================================================ # SECTION 6: BAIDU # ============================================================ User-agent: Baiduspider Allow: / User-agent: Baiduspider-image Allow: / User-agent: Baiduspider-video Allow: / User-agent: Baiduspider-news Allow: / # ============================================================ # SECTION 7: APPLE # ============================================================ # Applebot = Apple Search, Spotlight, Safari → ALLOW User-agent: Applebot Allow: / # Applebot-Extended = Apple Intelligence AI training → BLOCK User-agent: Applebot-Extended Disallow: / # ============================================================ # SECTION 8: CLOUDFLARE INTERNAL BOT # CloudflareBrowserRenderingCrawler: Cloudflare's own rendering # bot used for SEO diagnosis and page preview features. # Since you use Cloudflare, blocking this can break Cloudflare # features. Cloudflare's own template blocks it as a default # safety measure — but for your site Allow is correct. # ============================================================ User-agent: CloudflareBrowserRenderingCrawler Allow: / # ============================================================ # SECTION 9: HUAWEI / PETAL SEARCH # ============================================================ User-agent: PetalBot Allow: / # ============================================================ # SECTION 10: NAVER (South Korea) # ============================================================ User-agent: Yeti Allow: / # ============================================================ # SECTION 11: SEZNAM (Czech Republic) # ============================================================ User-agent: SeznamBot Allow: / # ============================================================ # SECTION 12: SOGOU (China) # ============================================================ User-agent: Sogou web spider Allow: / User-agent: Sogou Pic Spider Allow: / # ============================================================ # SECTION 13: BRAVE SEARCH # ============================================================ User-agent: Brave Allow: / # ============================================================ # SECTION 14: YOU.COM AI SEARCH # ============================================================ User-agent: YouBot Allow: / # ============================================================ # SECTION 15: OPENAI / CHATGPT # Source: OpenAI official documentation (Dec 2025) # ============================================================ # GPTBot = OpenAI model TRAINING → BLOCK # No referral traffic, only harvests training data User-agent: GPTBot Disallow: / # OAI-SearchBot = ChatGPT Search citations → ALLOW # Powers real-time web citations in ChatGPT Search User-agent: OAI-SearchBot Allow: / # ChatGPT-User = User-triggered live browsing → ALLOW User-agent: ChatGPT-User Allow: / # ============================================================ # SECTION 16: ANTHROPIC / CLAUDE # Source: Anthropic official documentation (Feb 2026) # anthropic-ai and claude-web are RETIRED — not included # ============================================================ # ClaudeBot = Anthropic model TRAINING → BLOCK User-agent: ClaudeBot Disallow: / # Claude-User = User-triggered real-time fetch → ALLOW # Blocking removes your site from live Claude citations User-agent: Claude-User Allow: / # Claude-SearchBot = Claude search quality index → ALLOW User-agent: Claude-SearchBot Allow: / # ============================================================ # SECTION 17: PERPLEXITY AI # ============================================================ User-agent: PerplexityBot Allow: / User-agent: Perplexity-User Allow: / # ============================================================ # SECTION 18: MISTRAL AI # ============================================================ # MistralAI-User = user-triggered retrieval, NOT training User-agent: MistralAI-User Allow: / # ============================================================ # SECTION 19: AMAZON (Alexa / Rufus / Nova) # Source: Amazon official documentation (June 2026) # Amazon now fully honors robots.txt as of June 15, 2026 # Cloudflare default BLOCKS Amazonbot — we OVERRIDE to Allow # Reason: Alexa + Rufus citation value for Sarkari Results # To opt out of Nova AI training only (without blocking): # Add per page # ============================================================ User-agent: Amazonbot Allow: / # Amzn-SearchBot = Alexa search results (verified NOT training) User-agent: Amzn-SearchBot Allow: / # Amzn-User = User action support (verified NOT training) User-agent: Amzn-User Allow: / # ============================================================ # SECTION 20: META / FACEBOOK # ============================================================ # Meta AI training bots → ALL BLOCKED User-agent: FacebookBot Disallow: / User-agent: meta-externalagent Disallow: / User-agent: Meta-ExternalAgent Disallow: / User-agent: Meta-ExternalFetcher Disallow: / # Facebook/Instagram link preview → ALLOW User-agent: facebookexternalhit Allow: / User-agent: Facebot Allow: / # ============================================================ # SECTION 21: SOCIAL MEDIA PREVIEW BOTS # ============================================================ User-agent: Twitterbot Allow: / User-agent: LinkedInBot Allow: / User-agent: Pinterestbot Allow: / User-agent: Instagram Allow: / User-agent: TelegramBot Allow: / # ============================================================ # SECTION 22: SEO / RESEARCH TOOLS # ============================================================ User-agent: SemrushBot Allow: / User-agent: AhrefsBot Allow: / User-agent: MJ12bot Allow: / User-agent: DotBot Allow: / User-agent: DataForSeoBot Allow: / # ============================================================ # SECTION 23: AI TRAINING & DATA HARVESTING — ALL BLOCKED # Only bots verified as training/harvesting with zero # search or citation value # ============================================================ # Common Crawl — backbone of almost all LLM training datasets User-agent: CCBot Disallow: / # Allen Institute for AI — research training datasets User-agent: AI2Bot Disallow: / # Cohere — AI model training User-agent: cohere-ai Disallow: / # ByteDance / TikTok — training crawler, no referral traffic User-agent: Bytespider Disallow: / # Diffbot — structured data harvesting for AI knowledge graphs User-agent: Diffbot Disallow: / # Omgili / Webz.io — AI training dataset aggregation User-agent: omgili Disallow: / User-agent: omgilibot Disallow: / # Generic scraping framework User-agent: Scrapy Disallow: / # ============================================================ # SECTION 24: DEFAULT RULE # ============================================================ User-agent: * Allow: /wp-admin/admin-ajax.php Disallow: /wp-admin/ Disallow: /cgi-bin/ Disallow: /*?s= Disallow: /*?utm_source= # /feed/ intentionally NOT blocked # RSS feed required for Google News + Google Discover # ============================================================ # SITEMAP — Rank Math # ============================================================ Sitemap: https://sarkariresult.com.ht/sitemap_index.xml # ============================================================ # END OF FILE # ============================================================