# robots.txt for www.marquardt.com # Block specific bots User-agent: GPTBot User-agent: BLEXBot User-agent: CloudVertexBot User-agent: ClaudeBot User-agent: Omgili User-agent: Omgilibot User-agent: Webzio-Extended User-agent: CCBot User-agent: FacebookBot User-agent: meta-externalagent User-agent: DeepseekBot User-agent: Applebot-Extended User-agent: Bytespider User-agent: cohere-training-data-crawler User-agent: PanguBot User-agent: Timpibot User-agent: AI2Bot User-agent: Diffbot User-agent: Amazonbot User-agent: AwarioRssBot User-agent: AwarioSmartBot User-agent: ImagesiftBot User-agent: Magpie-crawler User-agent: Peer39_crawler User-agent: Peer39_crawler/1.0 User-agent: MicrosoftPreview User-agent: Baiduspider User-agent: Baiduspider-image User-agent: Sogou web spider User-agent: Sogou Push Spider User-agent: Sogou Orion spider User-agent: Sogou inst spider User-agent: Sogou head spider User-agent: Sogou Pic Spider User-agent: Sogou spider2 User-agent: 360Spider User-agent: HaosouSpider User-agent: YisouSpider User-agent: YoudaoBot User-agent: EtaoSpider Disallow: / # Allow all others User-agent: * # Only allow URLs generated with frontend routing Disallow: /*?id=* Disallow: /*&id=* # L=0 is the default language Disallow: /*?L=0* Disallow: /*&L=0* # Should always be protected, but you know... Disallow: /*/Private/* Disallow: /*/Configuration/* # Disallow all files in /typo3temp/var/ Disallow: /typo3temp/var/* # Disallow all files in /typo3/ Disallow: /typo3/ # SQL-Dateien (Endungen) Disallow: /*.sql$ Disallow: /*.sql.gz$ # Prevent indexing of the 404 error page Disallow: /404 # Allow main newsroom page Allow: /newsroom$ Allow: /newsroom/detail/ Allow: /de/newsroom/detail/ # Disallow all filtered URLs that contain commas Disallow: /newsroom/*,* Disallow: /de/newsroom/*,* # Disallow news categories filter Disallow: /*categories=* # sitemap files Sitemap: https://www.marquardt.com/sitemap.xml