# See http://www.robotstxt.org/robotstxt.html for documentation on how to use the robots.txt file # Allow crawling of main pages User-agent: * Allow: / Allow: /play Allow: /about Allow: /books Allow: /educators Allow: /contact # Disallow crawling of filtered/parameterized exercise pages # These create infinite crawl paths and overwhelm the server Disallow: /play? Disallow: /en/play? Disallow: /fi/play? Disallow: /de/play? Disallow: /ja/play? Disallow: /nl/play? Disallow: /pl/play? Disallow: /it/play? Disallow: /lt/play? Disallow: /lv/play? Disallow: /ua/play? Disallow: /dk/play? Disallow: /no/play? Disallow: /ro/play? # Disallow admin area Disallow: /admin # Rate limit aggressive crawlers User-agent: Baiduspider Crawl-delay: 10 Disallow: /play? Disallow: /en/play? User-agent: Applebot Crawl-delay: 5 Disallow: /play? Disallow: /en/play? User-agent: AhrefsBot Crawl-delay: 10 User-agent: SemrushBot Crawl-delay: 10 User-agent: MJ12bot Crawl-delay: 10