# ============================================================== # robots.txt for technogeekscs.com # Django application | Last updated: 2026-06-17 # Optimized for: SEO crawlers + AI engines # (ChatGPT/GPTBot, Gemini/Google-Extended, Claude/ClaudeBot, # Copilot/Bingbot, Perplexity/PerplexityBot, CCBot + more) # ============================================================== # -------------------------------------------------------------- # SECTION 1: Default — allow all well-behaved crawlers # Block Django-specific internal/admin paths only # -------------------------------------------------------------- User-agent: * Allow: / # Django admin (never expose to crawlers) Disallow: /admin/ Disallow: /admin/login/ Disallow: /admin/logout/ Disallow: /admin/password_change/ # Django static/media internals (allow uploads folder separately below) Disallow: /static/admin/ Disallow: /media/private/ # Django REST framework / API endpoints (thin/machine content) Disallow: /api/ Disallow: /api/v1/ Disallow: /api/v2/ # Django auth & session URLs Disallow: /accounts/login/ Disallow: /accounts/logout/ Disallow: /accounts/password/ Disallow: /accounts/register/ Disallow: /accounts/signup/ # Django debug / utility paths Disallow: /__debug__/ Disallow: /debug/ Disallow: /silk/ Disallow: /rosetta/ # Search result pages (thin/duplicate content) Disallow: /search/ Disallow: /?q= Disallow: /?s= Disallow: /?search= Disallow: /?page= Disallow: /?sort= Disallow: /?filter= Disallow: /?category= Disallow: /?tag= # User account / transactional pages Disallow: /checkout/ Disallow: /cart/ Disallow: /my-account/ Disallow: /my-profile/ Disallow: /thank-you/ Disallow: /payment/ Disallow: /order/ Disallow: /download/ # Utility / housekeeping Disallow: /cgi-bin/ Disallow: /.well-known/ Disallow: /error/ Disallow: /404/ Disallow: /500/ # Always explicitly allow these critical files Allow: /robots.txt Allow: /llms.txt Allow: /sitemap.xml Allow: /static/images/ Allow: /static/css/ Allow: /static/js/ # -------------------------------------------------------------- # SECTION 2: Google (Search Index + Gemini AI) # -------------------------------------------------------------- User-agent: Googlebot Allow: / Disallow: /admin/ Allow: /llms.txt Allow: /sitemap.xml User-agent: Googlebot-Image Allow: /static/images/ # Google's dedicated Gemini / AI training crawler # This is separate from Googlebot — must be listed explicitly User-agent: Google-Extended Allow: / Allow: /llms.txt Allow: /courses/ Allow: /blog/ Allow: /about-us/ Allow: /placements/ Allow: /faq/ Disallow: /admin/ # -------------------------------------------------------------- # SECTION 3: ChatGPT / OpenAI # -------------------------------------------------------------- User-agent: GPTBot Allow: / Allow: /llms.txt Allow: /courses/ Allow: /blog/ Allow: /about-us/ Allow: /placements/ Allow: /faq/ Allow: /testimonials/ Disallow: /admin/ # ChatGPT browsing agent User-agent: ChatGPT-User Allow: / Allow: /llms.txt Disallow: /admin/ # OpenAI search bot User-agent: OAI-SearchBot Allow: / Allow: /llms.txt Disallow: /admin/ # -------------------------------------------------------------- # SECTION 4: Claude / Anthropic # -------------------------------------------------------------- User-agent: ClaudeBot Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: Claude-Web Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: anthropic-ai Allow: / Allow: /llms.txt Disallow: /admin/ # -------------------------------------------------------------- # SECTION 5: Microsoft Copilot / Bing # Copilot's knowledge comes entirely from Bing's index # -------------------------------------------------------------- User-agent: Bingbot Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: msnbot Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: BingPreview Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: msnbot-media Allow: /static/images/ # -------------------------------------------------------------- # SECTION 6: Perplexity AI # -------------------------------------------------------------- User-agent: PerplexityBot Allow: / Allow: /llms.txt Disallow: /admin/ # -------------------------------------------------------------- # SECTION 7: Common Crawl # Used by many LLM training pipelines (GPT base models, etc.) # -------------------------------------------------------------- User-agent: CCBot Allow: / Allow: /llms.txt Disallow: /admin/ # -------------------------------------------------------------- # SECTION 8: Other AI / LLM crawlers (2025–2026) # -------------------------------------------------------------- # Meta AI (Llama model training) User-agent: meta-externalagent Allow: / Allow: /llms.txt Disallow: /admin/ User-agent: FacebookBot Allow: / Allow: /llms.txt Disallow: /admin/ # Apple Applebot (Siri / Apple Intelligence) User-agent: Applebot Allow: / Allow: /llms.txt Disallow: /admin/ # Cohere AI User-agent: cohere-ai Allow: / Allow: /llms.txt Disallow: /admin/ # Amazon (Alexa / Amazonbot) User-agent: Amazonbot Allow: / Allow: /llms.txt Disallow: /admin/ # Bytedance / Bytespider User-agent: Bytespider Allow: / Allow: /llms.txt Disallow: /admin/ # Diffbot (AI knowledge graph) User-agent: Diffbot Allow: / Allow: /llms.txt Disallow: /admin/ # You.com AI User-agent: YouBot Allow: / Allow: /llms.txt Disallow: /admin/ # iAsk AI User-agent: iaskspider/2.0 Allow: / Allow: /llms.txt Disallow: /admin/ # Timpibot (used by Exa AI) User-agent: Timpibot Allow: / Allow: /llms.txt Disallow: /admin/ # Webz.io / Webhose (AI data feeds) User-agent: SemrushBot-SI Disallow: / # LinkedIn crawler (for social preview) User-agent: LinkedInBot Allow: / Allow: /about-us/ Disallow: /admin/ # WhatsApp / Facebook link preview User-agent: facebookexternalhit Allow: / Disallow: /admin/ # Twitter/X card preview User-agent: Twitterbot Allow: / Disallow: /admin/ # -------------------------------------------------------------- # SECTION 9: Block bad bots — SEO scrapers & spam crawlers # These waste your server's crawl budget with zero benefit # -------------------------------------------------------------- User-agent: AhrefsBot Disallow: / User-agent: MJ12bot Disallow: / User-agent: DotBot Disallow: / User-agent: SemrushBot Disallow: / User-agent: MajesticSEO Disallow: / User-agent: BLEXBot Disallow: / User-agent: DataForSeoBot Disallow: / User-agent: PetalBot Disallow: / User-agent: SeznamBot Disallow: / User-agent: Sogou Disallow: / User-agent: ia_archiver Disallow: / User-agent: EmailCollector Disallow: / User-agent: EmailSiphon Disallow: / User-agent: WebCopier Disallow: / User-agent: HTTrack Disallow: / User-agent: WebStripper Disallow: / User-agent: WebSauger Disallow: / User-agent: Offline Explorer Disallow: / User-agent: TeleportPro Disallow: / User-agent: harvest Disallow: / User-agent: SurveyBot Disallow: / User-agent: NutritionSearchBot Disallow: / User-agent: 360Spider Disallow: / User-agent: YandexBot Disallow: / # -------------------------------------------------------------- # SECTION 10: Crawl delay hints for heavy crawlers # (Not honoured by Google/Bing but respected by others) # -------------------------------------------------------------- User-agent: CCBot Crawl-delay: 10 User-agent: Diffbot Crawl-delay: 5 User-agent: Bytespider Crawl-delay: 10 # -------------------------------------------------------------- # SECTION 11: Sitemap declarations # List every sitemap — crawlers discover them from here # -------------------------------------------------------------- Sitemap: https://technogeekscs.com/sitemap.xml # LLMs.txt location — canonical AI knowledge file # LLMs-Txt: https://technogeekscs.com/llms.txt # ============================================================== # END OF robots.txt — technogeekscs.com # Maintained by: contact@technogeekscs.co.in # Django app | Review & update every 3 months # ==============================================================