NeuralCrawl

X (Twitter) / robots.txt snapshot

← back to x.com · fetched 2026-07-07T02:22:00Z (30d ago) · HTTP 200 · 2932 bytes · sha256 ca89b3e4df43728a · raw

final URL: https://x.com/robots.txt

1# Google Search Engine Robot
2# ==========================
3User-agent: Googlebot
4
5Allow: /*?s=
6Allow: /*?t=
7Allow: /*?ref_src=
8Allow: /hashtag/*?src=
9Allow: /search?q=%23
10Allow: /i/api/
11# Same length as the /*/likes and /*/media rules; RFC 9309 breaks the tie in
12# favor of Allow, so /i/api/ fetches stay crawlable.
13Allow: /i/api/*
14Disallow: /*?lang=en-ss
15Allow: /*?lang=
16Disallow: /search/realtime
17Disallow: /search/users
18Disallow: /search/*/grid
19
20Disallow: /*/analytics
21Disallow: /*/followers
22Disallow: /*/following
23Disallow: /*/verified_followers
24
25Disallow: /account/deactivated
26Disallow: /settings/deactivated
27
28# Only `*` and `$` are metacharacters here (RFC 9309); regex classes match
29# literally. /photo is anchored so /status/*/photo/1 media pages stay open.
30Disallow: /*/status/*/likes
31Disallow: /*/status/*/retweets
32Disallow: /*/likes
33Disallow: /*/likes?
34Disallow: /*/media
35Disallow: /*/media?
36Disallow: /*/photo$
37Disallow: /*/photo?
38Allow: /*?
39
40User-agent: Bingbot
41
42Allow: /*?s=
43Allow: /*?t=
44Allow: /*?ref_src=
45Allow: /hashtag/*?src=
46Allow: /search?q=%23
47Allow: /i/api/
48Allow: /i/api/*
49Disallow: /*?lang=en-ss
50Allow: /*?lang=
51Disallow: /search/realtime
52Disallow: /search/users
53Disallow: /search/*/grid
54
55Disallow: /*/analytics
56Disallow: /*/followers
57Disallow: /*/following
58Disallow: /*/verified_followers
59
60Disallow: /account/deactivated
61Disallow: /settings/deactivated
62
63Disallow: /*/status/*/likes
64Disallow: /*/status/*/retweets
65Disallow: /*/likes
66Disallow: /*/likes?
67Disallow: /*/media
68Disallow: /*/media?
69Disallow: /*/photo$
70Disallow: /*/photo?
71Allow: /*?
72
73User-agent: facebookexternalhit
74
75Allow: /*?lang=
76Allow: /*?s=
77Allow: /*?t=
78Allow: /*?ref_src=
79Allow: /hashtag/*?src=
80Allow: /search?q=%23
81Allow: /search?*cashtagRestId=
82Allow: /i/api/
83Allow: /i/api/*
84Disallow: /search/realtime
85Disallow: /search/users
86Disallow: /search/*/grid
87
88Disallow: /*?
89Disallow: /*/followers
90Disallow: /*/following
91Disallow: /*/verified_followers
92
93Disallow: /account/deactivated
94Disallow: /settings/deactivated
95
96Disallow: /*/status/*/likes
97Disallow: /*/status/*/retweets
98Disallow: /*/likes
99Disallow: /*/likes?
100Disallow: /*/media
101Disallow: /*/media?
102Disallow: /*/photo$
103Disallow: /*/photo?
104
105User-Agent: Google-Extended
106Disallow: *
107
108User-Agent: FacebookBot
109Disallow: *
110
111User-agent: Discordbot
112Disallow: *
113
114# Every bot that might possibly read and respect this file
115# ========================================================
116User-agent: *
117Disallow: /
118
119
120# WHAT-4882 - Keep notification-email links (/i/u) out of search results.
121# Named crawlers stay un-blocked on purpose: they must crawl /i/u to see its
122# X-Robots-Tag noindex (the robots.txt Noindex directive died in 2019).
123Disallow: /i/u
124
125# Wait 1 second between successive requests. See ONBOARD-2698 for details.
126Crawl-delay: 1
127
128# Independent of user agent. Links in the sitemap are full URLs using https://
129# and need to match the protocol of the sitemap.
130Sitemap: https://x.com/sitemap.xml