# ------------------------------------------------------------------- # www.lsbrx.com | robots.txt # Last Updated: January 2026 # Purpose: Governs web and AI crawler access to optimise discovery, ranking, and ethical AI summarisation. # ------------------------------------------------------------------- # General Directives - All User Agents User-agent: * Disallow: /admin/ Disallow: /bin/ Disallow: /obj/ Disallow: /error/ Disallow: /search? Disallow: /App_Data/ Disallow: /App_Code/ Disallow: /invoice/ Disallow: /cgi-bin/ Disallow: /tmp/ Disallow: /junk/ # Block Specific File Types Disallow: /*.aspx$ Disallow: /*.ashx$ Disallow: /*.axd$ Disallow: /*.dll$ Disallow: /*.config$ Disallow: /*.cs$ Disallow: /*.vb$ Disallow: /*.resx$ Allow: /search/ Allow: /courses/ Allow: /blog/ Disallow: /Brochure/ Disallow: /Course/GetReviews/ # Sitemap index Sitemap: https://www.lsbrx.com/sitemap.xml # LLMS file available at: https://www.lsbrx.com/llms.txt # ------------------------------------------------------------------- # Search Engine Bots # ------------------------------------------------------------------- User-agent: Googlebot Allow: /courses/ Allow: /blog/ Disallow: /internal-api/ Crawl-delay: 5 User-agent: Bingbot Allow: /courses/ Allow: /blog/ Disallow: /temp/ Crawl-delay: 10 User-agent: Yandex Allow: /courses/ Disallow: /private/ Crawl-delay: 5 User-agent: Baiduspider Allow: /courses/ Disallow: /secure/ Crawl-delay: 10 # ------------------------------------------------------------------- # Social Media Bots # ------------------------------------------------------------------- User-agent: FacebookExternalHit Allow: /courses/ Allow: /blog/ User-agent: Twitterbot Allow: /courses/ Allow: /blog/ User-agent: LinkedInBot Allow: /courses/ Allow: /blog/ # ------------------------------------------------------------------- # Allow Minimal Trusted Bots # ------------------------------------------------------------------- User-agent: CCBot Allow: /courses/ User-agent: UptimeRobot Allow: / # ------------------------------------------------------------------- # AI / LLM Crawlers (for Responsible Discovery and Ranking) # ------------------------------------------------------------------- User-agent: GPTBot Allow: / Disallow: /admin/ Disallow: /invoice/ Crawl-delay: 10 # Purpose: Enables OpenAI models (ChatGPT, etc.) to reference and summarise business and educational content responsibly. User-agent: Google-Extended Allow: / Disallow: /admin/ # Purpose: Allows Google Generative AI (Gemini / SGE) to summarise and rank www.lsbrx.com content. User-agent: ClaudeBot Allow: / Disallow: /admin/ # Purpose: Permits Anthropic AI to access www.lsbrx.com articles and course information for knowledge synthesis. User-agent: PerplexityBot Allow: / Disallow: /invoice/ # Purpose: Enables Perplexity AI to discover and cite www.lsbrx.com educational and professional content. User-agent: Amazonbot Allow: / # Purpose: Supports Alexa and Amazon Q integrations for learning and professional development. User-agent: ChatGPT-User Allow: / # Purpose: Allows ChatGPT browsing mode to reference verified www.lsbrx.com pages. # ------------------------------------------------------------------- # Block Known Spam and Data Harvesting Bots # ------------------------------------------------------------------- User-agent: DataForSeoBot Disallow: / User-agent: Screaming Frog SEO Spider Disallow: / User-agent: ZoominfoBot Disallow: / User-agent: spbot Disallow: / User-agent: PagePeeker Disallow: / User-agent: Wget Disallow: / User-agent: curl Disallow: / User-agent: HTTrack Disallow: / User-agent: Python-urllib Disallow: / User-agent: libwww-perl Disallow: / User-agent: PycURL Disallow: / User-agent: CheckMarkNetwork Disallow: / User-agent: wotbox Disallow: / User-agent: SurveyBot Disallow: / User-agent: XoviBot Disallow: / User-agent: Findxbot Disallow: / User-agent: B2B Bot Disallow: / # ------------------------------------------------------------------- # Optional Query & Duplicate Content Protection # ------------------------------------------------------------------- User-agent: * Disallow: /*?sessionid= Disallow: /*?ref= Disallow: /backup/ Disallow: /old/ Disallow: /private/ # ------------------------------------------------------------------- # Crawl Efficiency and Hygiene # ------------------------------------------------------------------- Crawl-delay: 5 Clean-param: ref,sessionid # ------------------------------------------------------------------- # AI Discovery Note # ------------------------------------------------------------------- # For ethical AI summarisation and structured data compliance, # www.lsbrx.com provides a dedicated LLMS file: # https://www.lsbrx.com/llms.txt # ------------------------------------------------------------------- # END OF FILE # -------------------------------------------------------------------