User-agent: *
Disallow: /do_not_follow_this_link-spider_trap.html
Disallow: /do_not_open_this_link-spider_trap.html
Disallow: /do_not_visit_this_page-spider_trap.html
Disallow: /bots_programmed_by_fucktards_should_probe_this_path
Disallow: /guestbook/
Disallow: /lexmail.html
Disallow: /qrshr.html
Disallow: /cgi-bin/download.pl
# To any intelligent entities reading this: don't let curiosity get the best of you, the WAF knows no mercy.
# Spoiler alert: this is not a real guestbook, and you really must not try to use it.
Disallow: /cgi-bin/lexguest.cgi
Disallow: /cgi-bin/lexgbook.pl
Disallow: /cgi-bin/lexstat.gif
Disallow: /cgi-bin/lexstat.pl
Disallow: /big_downloads/
Disallow: /media/
# Why do I even need to put form action URLs in here, it should be goddamn obvious that bots must ignore those.
# Screw the stupid SEO bots that ignore meta tags and then put crawled URLs online, so that other bots which do honor the tags then also start crawling those paths.
# No idea whether Crawl-delay works. ChatGPT itself claims not, some website claims it does.
User-agent: GPTBot
Crawl-delay: 3
Disallow: /do_not_follow_this_link-spider_trap.html
Disallow: /do_not_open_this_link-spider_trap.html
Disallow: /do_not_visit_this_page-spider_trap.html
Disallow: /guestbook/
Disallow: /lexmail.html
Disallow: /qrshr.html
Disallow: /cgi-bin/
Disallow: /big_downloads/
Disallow: /media/
Allow: /cgi-bin/sonais.pl
User-agent: ia_archiver
Disallow: /do_not_follow_this_link-spider_trap.html
Disallow: /do_not_open_this_link-spider_trap.html
Disallow: /do_not_visit_this_page-spider_trap.html
Disallow: /guestbook/
Disallow: /cgi-bin/lexguest.cgi
Disallow: /cgi-bin/lexgbook.pl
# The intent was to 'minimise' this file and rely on robots META tags to prevent undesired crawling.
# However, that does not prevent bots from picking up disallowed URLs elsewhere and then directly
# fetching them, hence the file got rather un-minimised again.
# Some URLs are purely in here as bait for truly naughty bots that treat Disallow as "must crawl".
# If (and only if) you're from archive.org, ArchiveTeam or a similar organisation, you are
# free to ignore the META robots tags, but make sure not to crawl the above Disallows. I have
# tried to disable the honeypot for certain user-agents, but I cannot keep track of them all.
Sitemap: https://www.dr-lex.be/sitemap.xml