# robots.txt - greekstatemuseum.com / kmst.gr # Goal: let crawlers index real content, but stop the parametric # collection-search "crawl trap" that hammers the DB with thousands # of ?param combinations and pagination (start=, show=, byArtist...). User-agent: * # Block the collections DB search with any query string (crawl trap) Disallow: /kmst/collections/db/search.html? # Block deep pagination on listing pages Disallow: /*?start= Disallow: /*&start= Disallow: /*?show= Disallow: /*&show= # Block internal / non-content endpoints Disallow: /kmst/register.html Disallow: /kmst/login.html Disallow: /*/cache/ # Allow the rest (activities, exhibitions, pressroom, static files, images) Allow: / # Be gentle - these are DB-backed pages Crawl-delay: 10