Der Spiegel / robots.txt snapshot
← back to spiegel.de · fetched 2026-08-11T15:08:50Z (1mo ago) · HTTP 200 · 5613 bytes · sha256 b1488648bb98cb62 · raw
final URL: https://www.spiegel.de/robots.txt
| 1 | User-agent: * |
| 2 | Allow: / |
| 3 | Disallow: /*CR-Dokumentation.pdf$ |
| 4 | |
| 5 | User-agent: emetriqContextualBot |
| 6 | Allow: / |
| 7 | |
| 8 | User-agent: Mozilla/5.0 (compatible; OGDWCtxCrawler) |
| 9 | Allow: / |
| 10 | |
| 11 | User-agent: AmazonAdBot |
| 12 | Allow: / |
| 13 | |
| 14 | # TLP-6507: Testweise Freischaltung der OpenAI-Suchcrawler fuer ausgewaehlte Bereiche |
| 15 | User-agent: OAI-SearchBot |
| 16 | Allow: /ausland/ |
| 17 | Allow: /partnerschaft/ |
| 18 | Allow: /gesundheit/ |
| 19 | Allow: /familie/ |
| 20 | Allow: /reise/ |
| 21 | Allow: /psychologie/ |
| 22 | Allow: /stil/ |
| 23 | Allow: /tests/ |
| 24 | Disallow: / |
| 25 | |
| 26 | # TLP-6507: Testweise Freischaltung der OpenAI-Suchcrawler fuer ausgewaehlte Bereiche |
| 27 | User-agent: ChatGPT-User |
| 28 | Allow: /ausland/ |
| 29 | Allow: /partnerschaft/ |
| 30 | Allow: /gesundheit/ |
| 31 | Allow: /familie/ |
| 32 | Allow: /reise/ |
| 33 | Allow: /psychologie/ |
| 34 | Allow: /stil/ |
| 35 | Allow: /tests/ |
| 36 | Disallow: / |
| 37 | |
| 38 | # ========================================== |
| 39 | # KI-CRAWLER & KI-TRAINING (SPERRE) |
| 40 | # ========================================== |
| 41 | User-agent: GPTBot |
| 42 | Disallow: / |
| 43 | User-agent: anthropic-ai |
| 44 | Disallow: / |
| 45 | User-agent: Claude-Web |
| 46 | Disallow: / |
| 47 | User-agent: ClaudeBot |
| 48 | Disallow: / |
| 49 | User-agent: Claude-SearchBot |
| 50 | Disallow: / |
| 51 | User-agent: Claude-User |
| 52 | Disallow: / |
| 53 | User-agent: CloudVertexBot |
| 54 | Disallow: / |
| 55 | User-agent: cohere-ai |
| 56 | Disallow: / |
| 57 | User-agent: cohere-training-data-crawler |
| 58 | Disallow: / |
| 59 | User-agent: cohere-training-data-collector |
| 60 | Disallow: / |
| 61 | User-agent: DeepSeekBot |
| 62 | Disallow: / |
| 63 | User-agent: DeepSeek |
| 64 | Disallow: / |
| 65 | User-agent: Meta-ExternalAgent |
| 66 | Disallow: / |
| 67 | User-agent: FacebookBot |
| 68 | Disallow: / |
| 69 | User-agent: Applebot-Extended |
| 70 | Disallow: / |
| 71 | User-agent: MistralAI-User |
| 72 | Disallow: / |
| 73 | User-agent: Amazonbot |
| 74 | Disallow: / |
| 75 | User-agent: AI2Bot |
| 76 | Disallow: / |
| 77 | User-agent: Kangaroo Bot |
| 78 | Disallow: / |
| 79 | User-agent: PanguBot |
| 80 | Disallow: / |
| 81 | User-agent: PetalBot |
| 82 | Disallow: / |
| 83 | User-agent: Devin |
| 84 | Disallow: / |
| 85 | User-agent: bigsur.ai |
| 86 | Disallow: / |
| 87 | User-agent: LinerBot |
| 88 | Disallow: / |
| 89 | User-agent: CCBot |
| 90 | Disallow: / |
| 91 | User-agent: YouBot |
| 92 | Disallow: / |
| 93 | User-agent: YouBot-Search |
| 94 | Disallow: / |
| 95 | User-agent: iaskspider |
| 96 | Disallow: / |
| 97 | User-agent: YoudaoBot |
| 98 | Disallow: / |
| 99 | User-agent: DeepL-Translate |
| 100 | Disallow: / |
| 101 | User-agent: DeepL-Bot |
| 102 | Disallow: / |
| 103 | |
| 104 | # ========================================== |
| 105 | # SEO-CRAWLER (SPERRE) |
| 106 | # ========================================== |
| 107 | User-agent: AhrefsBot |
| 108 | Disallow: / |
| 109 | User-agent: AhrefsSiteAudit |
| 110 | Disallow: / |
| 111 | User-agent: SemrushBot |
| 112 | Disallow: / |
| 113 | User-agent: SemrushBot-SA |
| 114 | Disallow: / |
| 115 | User-agent: SemrushBot-BA |
| 116 | Disallow: / |
| 117 | User-agent: SemrushBot-SI |
| 118 | Disallow: / |
| 119 | User-agent: SemrushBot-SWA |
| 120 | Disallow: / |
| 121 | User-agent: SiteAuditBot |
| 122 | Disallow: / |
| 123 | User-agent: SplitSignalBot |
| 124 | Disallow: / |
| 125 | User-agent: xovi |
| 126 | Disallow: / |
| 127 | User-agent: XoviBot |
| 128 | Disallow: / |
| 129 | User-agent: Seobility |
| 130 | Disallow: / |
| 131 | User-agent: SeobilityBot |
| 132 | Disallow: / |
| 133 | User-agent: RyteBot |
| 134 | Disallow: / |
| 135 | User-agent: SEOkicks |
| 136 | Disallow: / |
| 137 | User-agent: SEOkicks-Robot |
| 138 | Disallow: / |
| 139 | User-agent: Searchmetrics |
| 140 | Disallow: / |
| 141 | User-agent: SearchmetricsBot |
| 142 | Disallow: / |
| 143 | User-agent: audisto |
| 144 | Disallow: / |
| 145 | User-agent: audisto-essential |
| 146 | Disallow: / |
| 147 | User-agent: MJ12bot |
| 148 | Disallow: / |
| 149 | User-agent: dotbot |
| 150 | Disallow: / |
| 151 | User-agent: rogerbot |
| 152 | Disallow: / |
| 153 | User-agent: Ezooms |
| 154 | Disallow: / |
| 155 | User-agent: barkrowler |
| 156 | Disallow: / |
| 157 | User-agent: serpstatbot |
| 158 | Disallow: / |
| 159 | User-agent: BLEXBot |
| 160 | Disallow: / |
| 161 | User-agent: DataForSeoBot |
| 162 | Disallow: / |
| 163 | User-agent: MegaIndex |
| 164 | Disallow: / |
| 165 | User-agent: lumar |
| 166 | Disallow: / |
| 167 | User-agent: deepcrawl |
| 168 | Disallow: / |
| 169 | User-agent: oncrawl |
| 170 | Disallow: / |
| 171 | User-agent: sitebulb |
| 172 | Disallow: / |
| 173 | User-agent: botify |
| 174 | Disallow: / |
| 175 | User-agent: Siteimprove |
| 176 | Disallow: / |
| 177 | User-agent: Siteimprove Crawl |
| 178 | Disallow: / |
| 179 | User-agent: MozDotNet |
| 180 | Disallow: / |
| 181 | User-agent: Cocolyzebot |
| 182 | Disallow: / |
| 183 | User-agent: Raven |
| 184 | Disallow: / |
| 185 | |
| 186 | # ========================================== |
| 187 | # MEDIENBEOBACHTUNG, AGGREGATOREN & SCRAPER |
| 188 | # ========================================== |
| 189 | User-agent: Meltwater |
| 190 | Disallow: / |
| 191 | User-agent: NewsNow |
| 192 | Disallow: / |
| 193 | User-agent: Webzio-Extended |
| 194 | Disallow: / |
| 195 | User-agent: magpie-crawler |
| 196 | Disallow: / |
| 197 | User-agent: omgili |
| 198 | Disallow: / |
| 199 | User-agent: omgilibot |
| 200 | Disallow: / |
| 201 | User-agent: Baiduspider |
| 202 | Disallow: / |
| 203 | User-agent: Yeti |
| 204 | Disallow: / |
| 205 | User-agent: sentibot |
| 206 | Disallow: / |
| 207 | User-agent: Bytespider |
| 208 | Disallow: / |
| 209 | User-agent: SirdataBot |
| 210 | Disallow: / |
| 211 | User-agent: LCC |
| 212 | Disallow: / |
| 213 | User-agent: TurnitinBot |
| 214 | Disallow: / |
| 215 | User-agent: ImagesiftBot |
| 216 | Disallow: / |
| 217 | User-agent: Timpibot |
| 218 | Disallow: / |
| 219 | User-agent: Diffbot |
| 220 | Disallow: / |
| 221 | User-agent: Landau-Media-Spider |
| 222 | Disallow: / |
| 223 | |
| 224 | |
| 225 | # ========================================== |
| 226 | # HISTORISCHE BOTS & MASSEN-DOWNLOADER |
| 227 | # ========================================== |
| 228 | User-agent: Bloodhound |
| 229 | Disallow: / |
| 230 | User-agent: cydralspider |
| 231 | Disallow: / |
| 232 | User-agent: downloadexpress |
| 233 | Disallow: / |
| 234 | User-agent: gammaSpider |
| 235 | Disallow: / |
| 236 | User-agent: ObjectsSearch |
| 237 | Disallow: / |
| 238 | User-agent: Pimptrain |
| 239 | Disallow: / |
| 240 | User-agent: wapspider |
| 241 | Disallow: / |
| 242 | User-agent: WebZinger |
| 243 | Disallow: / |
| 244 | User-agent: Fasterfox |
| 245 | Disallow: / |
| 246 | |
| 247 | # ========================================== |
| 248 | # SCRAPING FRAMEWORKS & UTILITIES |
| 249 | # ========================================== |
| 250 | User-agent: Scrapy |
| 251 | Disallow: / |
| 252 | User-agent: HTTPBannerDetection |
| 253 | Disallow: / |
| 254 | User-agent: Wget |
| 255 | Disallow: / |
| 256 | |
| 257 | # Sitemaps |
| 258 | Sitemap: https://www.spiegel.de/sitemaps/news-de.xml |
| 259 | Sitemap: https://www.spiegel.de/sitemaps/videos/sitemap.xml |
| 260 | Sitemap: https://www.spiegel.de/plus/sitemap.xml |
| 261 | Sitemap: https://www.spiegel.de/sitemap.xml |
| 262 | |
| 263 | # Legal notice: spiegel.de expressly reserves the right to use its content for commercial text and data mining (§ 44b Urheberrechtsgesetz). |
| 264 | # The use of robots or other automated means to access spiegel.de or collect or mine data without the express permission of spiegel.de is strictly prohibited. |
| 265 | # spiegel.de may, in its discretion, permit certain automated access to certain spiegel.de pages, |
| 266 | # If you would like to apply for permission to crawl spiegel.de, collect or use data, please email [email protected] |
| 267 |