Robots.txt

Robots.txt Rules

Crawler Rules
All
  • Allow: /
  • Disallow: /user/
  • Disallow: /members/
  • Disallow: /members/newsletters/
  • Disallow: /members/alerts/add/
  • Disallow: /search/
  • Disallow: *Xhr
  • Disallow: */xhr
  • Disallow: */ajax/
  • Disallow: */fly/*/bundles/flyjs/
  • Disallow: */libs/
  • Disallow: */version!libs/
  • Disallow: /.well-known/
  • Disallow: /index.php/
  • Disallow: *?beta=
  • Disallow: *?ftag=
MSN
  • Allow: /
addsearchbot
  • Allow: /
ai2bot
  • Allow: /
ai2bot-deepresearcheval
  • Allow: /
ai2bot-dolma
  • Allow: /
aiwebindex
  • Allow: /
amazon-qbusiness
  • Allow: /
amazonbot
  • Allow: /
amazonbuyforme
  • Allow: /
amzn-searchbot
  • Allow: /
amzn-user
  • Allow: /
anchor browser
  • Allow: /
anthropic-ai
  • Allow: /
apifybot
  • Allow: /
apifywebsitecontentcrawler
  • Allow: /
applebot
  • Allow: /
applebot-extended
  • Allow: /
atlassian-bot
  • Allow: /
autorag
  • Allow: /
awariosmartbot
  • Allow: /
  • Allow: /
azureai-searchbot
  • Allow: /
big sur ai
  • Allow: /
bigsur.ai
  • Allow: /
brandwatch
  • Allow: /
bravebot
  • Allow: /
brightbot
  • Allow: /
bytespider
  • Allow: /
ccbot
  • Allow: /
channel3bot
  • Allow: /
chatglm-spider
  • Allow: /
chatgpt-user
  • Allow: /
claude-code
  • Allow: /
claude-searchbot
  • Allow: /
claude-user
  • Allow: /
claude-web
  • Allow: /
claudebot
  • Allow: /
code
  • Allow: /
cohere-ai
  • Allow: /
cohere-training-data-crawler
  • Allow: /
cotoyogi
  • Allow: /
crawl4ai
  • Allow: /
cursor
  • Allow: /
datenbank crawler
  • Allow: /
datenbank-crawler
  • Allow: /
deepseekbot
  • Allow: /
devin
  • Allow: /
diffbot
  • Allow: /
direqt anomura
  • Allow: /
duckassistbot
  • Allow: /
echobot bot
  • Allow: /
exabot
  • Allow: /
facebookbot
  • Allow: /
factset_spyderbot
  • Allow: /
firecrawlagent
  • Allow: /
geisthaus-pagefetcher
  • Allow: /
gptbot
  • Allow: /
iask
  • Allow: /
iaskbot
  • Allow: /
iaskspider
  • Allow: /
icc crawler
  • Allow: /
icc-crawler
  • Allow: /
imagesiftbot
  • Allow: /
imagespider
  • Allow: /
kagi-fetcher
  • Allow: /
kangaroo bot
  • Allow: /
kangaroo-bot
  • Allow: /
kimi-user
  • Allow: /
klaviyoaibot
  • Allow: /
kunato
  • Allow: /
laion-huggingface-processor
  • Allow: /
lcc
  • Allow: /
liner bot
  • Allow: /
linerbot
  • Allow: /
linkupbot
  • Allow: /
manus-user
  • Allow: /
meta-externalagent
  • Allow: /
meta-externalfetcher
  • Allow: /
meta-webindexer
  • Allow: /
mistral.ai
  • Allow: /
mistralai-user
  • Allow: /
netestate imprint crawler
  • Allow: /
novaact
  • Allow: /
novellum
  • Allow: /
novellum ai crawl
  • Allow: /
oai-searchbot
  • Allow: /
omgili
  • Allow: /
opencode
  • Allow: /
pangubot
  • Allow: /
peopleinc-dcipher-scraper/1.1.0
  • Allow: /
perplexity-user
  • Allow: /
perplexitybot
  • Allow: /
petalbot
  • Allow: /
phindbot
  • Allow: /
poggio-citations
  • Allow: /
qualifiedbot
  • Allow: /
sbintuitionsbot
  • Allow: /
seekrbot
  • Allow: /
semrushbot-ocob
  • Allow: /
semrushbotswa
  • Allow: /
shap-user
  • Allow: /
shapbot
  • Allow: /
spider
  • Allow: /
tavilybot
  • Allow: /
terracotta
  • Allow: /
timpibot
  • Allow: /
tongyibot
  • Allow: /
trae
  • Allow: /
twinagent
  • Allow: /
useai
  • Allow: /
velen crawler
  • Allow: /
velenpublicwebcrawler
  • Allow: /
webzio-extended
  • Allow: /
wrtn
  • Allow: /
yiyanbot
  • Allow: /
youbot
  • Allow: /
zanistabot
  • Allow: /
  • Allow: /article/how-ai-companies-are-secretly-collecting-training-data-from-the-web-and-why-it-matters/
  • Allow: /article/how-proxy-servers-actually-work-and-why-theyre-so-valuable/
  • Allow: /article/how-to-remove-your-personal-information-from-whitepages-in-5-steps-and-why-you-should/
  • Allow: /article/how-to-undo-a-reconciliation-in-quickbooks-online-the-easy-way/
  • Allow: /article/i-found-the-easiest-way-to-delete-myself-from-the-internet-and-why-you-shouldnt-wait-to-use-it-too/
  • Allow: /article/incogni-vs-deleteme/
  • Allow: /article/this-proxy-provider-i-tested-is-the-best-for-web-scraping-and-its-not-iproyal-or-marsproxies/
  • Disallow: /
"008"
  • Allow: /
awariorssbot
  • Allow: /
ddm-dcipher/1.0.7
  • Allow: /
httrack
  • Allow: /
magpie-crawler
  • Allow: /
nutch
  • Allow: /
offline explorer
  • Allow: /
omgilibot
  • Allow: /
peer39_crawler/1.0
  • Allow: /
scrapy
  • Allow: /
  • Disallow: /

Sitemaps

Priority Location
6https://www.zdnet.com/sitemaps/news.xml
5https://www.zdnet.com/sitemaps/topics.xml
4https://www.zdnet.com/sitemaps/article/index.xml
3https://www.zdnet.com/sitemaps/gallery/index.xml
2https://www.zdnet.com/sitemaps/video/index.xml
1https://www.zdnet.com/sitemaps/review/index.xml

Raw Text

# www.robotstxt.org/
# www.google.com/support/webmasters/bin/answer.py?hl=en&answer=156449

# Ziff Davis content is made available for your non-commercial use subject to our
# Terms of Use here: https://www.ziffdavis.com/terms-of-use
# Use of any robot, crawler, or other tool to scrape, harvest, extract, or retrieve any content on
# this website using automated means is prohibited without written permission from Ziff Davis.
# Prohibited uses include but are not limited to:
# (1) text and data mining under Art. 4 of the EU Directive on Copyright in the Digital Single
# Market;
# (2) development or operation of artificial intelligence or machine learning software or
# databases, including by training, fine-tuning, embedding, and retrieval-augmented generation;
# (3) creating data sets containing our content or sharing it with others; and
# (4) any commercial purposes.
# Contact licensing@ziffdavis.com for assistance.

User-agent: *

Disallow: /user/*
Disallow: /members/
Disallow: /members/newsletters/
Disallow: /members/alerts/add/
Disallow: /search/
Disallow: *Xhr*
Disallow: */xhr*
Disallow: */ajax/*
Disallow: */fly/*/bundles/flyjs/*
Disallow: */libs/*
Disallow: */version!libs/*
Disallow: /.well-known/*
Disallow: /index.php/*
Disallow: *?beta=*
Disallow: *?ftag=*

User-agent: msnbot
Crawl-delay: 1

# Ziff Davis bot list
user-agent: AddSearchBot
user-agent: AI2Bot
user-agent: Ai2Bot-DeepResearchEval
user-agent: Ai2Bot-Dolma
user-agent: AIWebIndex
user-agent: amazon-QBusiness
user-agent: Amazonbot
user-agent: AmazonBuyForMe
user-agent: Amzn-SearchBot
user-agent: Amzn-User
user-agent: Anchor Browser
user-agent: anthropic-ai
user-agent: ApifyBot
user-agent: ApifyWebsiteContentCrawler
user-agent: Applebot
user-agent: Applebot-extended
user-agent: atlassian-bot
user-agent: AutoRAG
user-agent: AwarioSmartBot
user-agent: AzureAI-SearchBot
user-agent: Big Sur AI
user-agent: bigsur.ai
user-agent: Brandwatch
user-agent: Bravebot
user-agent: Brightbot
user-agent: Bytespider
user-agent: CCBot
user-agent: Channel3Bot
user-agent: ChatGLM-Spider
user-agent: ChatGPT-User
user-agent: Claude-Code
user-agent: Claude-SearchBot
user-agent: Claude-User
user-agent: Claude-Web
user-agent: ClaudeBot
user-agent: Code
user-agent: cohere-ai
user-agent: cohere-training-data-crawler
user-agent: Cotoyogi
user-agent: Crawl4AI
user-agent: Cursor
user-agent: Datenbank Crawler
user-agent: Datenbank-Crawler
user-agent: DeepSeekBot
user-agent: Devin
user-agent: Diffbot
user-agent: Direqt Anomura
user-agent: DuckAssistBot
user-agent: Echobot Bot
user-agent: ExaBot
user-agent: FacebookBot
user-agent: Factset_spyderbot
user-agent: FirecrawlAgent
user-agent: GeistHaus-PageFetcher
user-agent: GPTBot
user-agent: iAsk
user-agent: iAskBot
user-agent: iaskspider
user-agent: ICC Crawler
user-agent: ICC-Crawler
user-agent: ImageSiftBot
user-agent: imageSpider
user-agent: kagi-fetcher
user-agent: Kangaroo Bot
user-agent: Kangaroo-Bot
user-agent: Kimi-User
user-agent: KlaviyoAIBot
user-agent: Kunato
user-agent: laion-huggingface-processor
user-agent: LCC
user-agent: LINER Bot
user-agent: LinerBot
user-agent: LinkupBot
user-agent: Manus-User
user-agent: meta-externalagent
user-agent: meta-externalfetcher
user-agent: meta-webindexer
user-agent: mistral.ai
user-agent: MistralAI-User
user-agent: netEstate Imprint Crawler
user-agent: NovaAct
user-agent: Novellum
user-agent: Novellum AI Crawl
user-agent: OAI-SearchBot
user-agent: omgili
user-agent: opencode
user-agent: PanguBot
user-agent: PeopleInc-DCipher-Scraper/1.1.0
user-agent: Perplexity-User
user-agent: PerplexityBot
user-agent: PetalBot
user-agent: PhindBot
user-agent: Poggio-Citations
user-agent: QualifiedBot
user-agent: SBIntuitionsBot
user-agent: SeekrBot
user-agent: SemrushBot-OCOB
user-agent: SemrushBotSwa
user-agent: Shap-User
user-agent: ShapBot
user-agent: Spider
user-agent: TavilyBot
user-agent: TerraCotta
user-agent: Timpibot
user-agent: TongyiBot
user-agent: Trae
user-agent: TwinAgent
user-agent: UseAI
user-agent: Velen Crawler
user-agent: VelenPublicWebCrawler
user-agent: Webzio-Extended
user-agent: Wrtn
user-agent: YiyanBot
user-agent: YouBot
user-agent: ZanistaBot
Allow: /article/how-ai-companies-are-secretly-collecting-training-data-from-the-web-and-why-it-matters/
Allow: /article/how-proxy-servers-actually-work-and-why-theyre-so-valuable/
Allow: /article/how-to-remove-your-personal-information-from-whitepages-in-5-steps-and-why-you-should/
Allow: /article/how-to-undo-a-reconciliation-in-quickbooks-online-the-easy-way/
Allow: /article/i-found-the-easiest-way-to-delete-myself-from-the-internet-and-why-you-shouldnt-wait-to-use-it-too/
Allow: /article/incogni-vs-deleteme/
Allow: /article/this-proxy-provider-i-tested-is-the-best-for-web-scraping-and-its-not-iproyal-or-marsproxies/
Disallow: /

#ZDnet restricted bots
User-agent: "008"
User-agent: AwarioRssBot
User-agent: AwarioSmartBot
User-agent: DDM-DCipher/1.0.7
User-agent: HTTrack
User-agent: Magpie-crawler
User-agent: Nutch
User-agent: Offline Explorer
User-agent: Omgilibot
User-agent: Peer39_crawler/1.0
User-agent: Scrapy
Disallow: /

Sitemap: https://www.zdnet.com/sitemaps/news.xml
Sitemap: https://www.zdnet.com/sitemaps/topics.xml
Sitemap: https://www.zdnet.com/sitemaps/article/index.xml
Sitemap: https://www.zdnet.com/sitemaps/gallery/index.xml
Sitemap: https://www.zdnet.com/sitemaps/video/index.xml
Sitemap: https://www.zdnet.com/sitemaps/review/index.xml