{
 "question": "What share of the most-visited websites disallow each AI crawler in robots.txt, and do their files treat AI training crawlers differently from AI search and user-triggered crawlers? The study measures crawl permission stated in robots.txt only; it does not measure crawling, indexing, retrieval, citation or visibility in AI answers.",
 "population": "Registrable domains ranked 1-10,000 in the Tranco list L5PZ4 whose /robots.txt returned HTTP 200 with a plain-text body (n=5,572).",
 "inclusion": "HTTP 200 robots.txt that is not an HTML page.",
 "exclusion": "Unreachable hosts, HTTP 4xx/5xx robots.txt, HTML served at /robots.txt (reported separately).",
 "calculations": [
  "A bot is 'blocked' when RFC 9309 evaluation of the root path '/' is disallowed for its product token (named group, else the * group, longest-match rule, allow wins ties).",
  "'Blocked by name' counts only files that name the token in a user-agent line.",
  "Percentages use the parseable-file denominator; rank-tier percentages use each tier's parseable files.",
  "95% intervals: Wilson score intervals for shares; Newcombe hybrid score intervals for differences between rank tiers.",
  "Rank gradient: Cochran-Armitage trend test over the three published tiers and over ten bands of 1,000 ranks, repeated on domains with a live homepage and on files that do not disallow the root for unnamed bots.",
  "Denominator sensitivity: the same counts over readable files plus 4xx responses (under RFC 9309 a missing file allows everything), over all 10,000 domains (a lower bound: unreadable files counted as not blocking), and over domains with a live homepage.",
  "Rule source: a block is 'by name' when a user-agent group names the token, and 'via wildcard' when no group names it and the * group disallows the root.",
  "Path restrictions: root allowed but at least one disallow rule applies, split by named vs wildcard group. 'Closed to the crawler, open to Googlebot' tests every concrete path the file itself lists in a disallow rule (text before the first * or $) for the crawler and for Googlebot.",
  "Crawler purposes (training, search, user-triggered) are the operators' published descriptions as of 2026-09-26; see crawler_taxonomy.",
  "Citation cross-check (1.2): sources cited on 2026-09-26 by ChatGPT, Claude, Gemini and Perplexity (80 buyer questions, s23_citations_vs_google.csv) and by 481 US AI Overviews (s5_ai_overview_citations.csv), deduplicated per engine, matched to top-10,000 domains by host suffix; the engine operator's own domains excluded. Page-level permission tests the cited URL's path and query for the engine's search crawler, only for URLs on the domain or its www host. Expected rates reweight the population share of blocking files to the cited domains' mix of 1,000-rank bands; exact binomial tests."
 ],
 "limitations": [
  "Tranco ranks domains, including infrastructure domains without a public website.",
  "Robots.txt expresses a request, not enforcement; firewall-level blocking (e.g. Cloudflare bot rules) is not measured.",
  "A single fetch per domain on one day; files change.",
  "Some servers treat an unknown user agent differently; we used one honest research user agent.",
  "Permission is not crawling, indexing, retrieval or citation: a site that allows a search crawler may still never appear in AI answers, and a site that disallows it may be reached through another index or a user-triggered fetch.",
  "Robots.txt settings show what files say, not why. A training crawler blocked by name while the search crawler is not mentioned is allowed by default, not necessarily by choice.",
  "Differences between rank tiers are associations. Popular domains differ in type, size and ownership, which this study does not control for.",
  "Root-level blocking is a coarse measure; path-level rules are summarized, but the study does not know which of a site's pages matter for AI answers.",
  "Citation cross-check: one day, a few hundred cited pages per assistant clustered on few domains; a citation shows a page was named, not how the engine obtained it."
 ],
 "update": "quarterly",
 "@context": "https://schema.org",
 "@type": "Dataset",
 "name": "Which AI crawlers do top websites block? 10,000 sites, 2026",
 "version": "1.2",
 "dateCreated": "2026-09-26",
 "creator": {
  "@type": "Organization",
  "name": "Underneath",
  "url": "https://underneath.agency"
 },
 "url": "https://underneath.agency/research/ai-crawler-blocking-study",
 "temporalCoverage": "2026-09-26",
 "isAccessibleForFree": true,
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "code": "cite/pipeline/ (collection, validation and analysis scripts)",
 "variableMeasured": [
  "sample",
  "s1_robots_state_all_attempted",
  "s1_robots_state_live_sites",
  "s1_n_parsed",
  "s1_per_bot",
  "s1_combos",
  "s1_any_ai_named_pct",
  "s1_any_ai_blocked_by_tier",
  "s1_tier_n",
  "s1_content_signal",
  "s1_robots_bytes_median",
  "s1_derived",
  "s1_content_signal_sites",
  "s1_v11",
  "s1_v12_linkage"
 ],
 "estimand": "Share of Tranco top-10,000 domains with an HTTP 200 plain-text robots.txt (n=5,572) whose file disallows the site root for a given product token under RFC 9309.",
 "crawler_taxonomy": {
  "as_of": "2026-09-26",
  "checked": "2026-09-28",
  "tokens": [
   {
    "token": "GPTBot",
    "operator": "OpenAI",
    "purpose": "training",
    "source": "https://platform.openai.com/docs/bots"
   },
   {
    "token": "OAI-SearchBot",
    "operator": "OpenAI",
    "purpose": "search",
    "source": "https://platform.openai.com/docs/bots"
   },
   {
    "token": "ChatGPT-User",
    "operator": "OpenAI",
    "purpose": "user",
    "source": "https://platform.openai.com/docs/bots"
   },
   {
    "token": "ClaudeBot",
    "operator": "Anthropic",
    "purpose": "training",
    "source": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
   },
   {
    "token": "Claude-SearchBot",
    "operator": "Anthropic",
    "purpose": "search",
    "source": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
   },
   {
    "token": "Claude-User",
    "operator": "Anthropic",
    "purpose": "user",
    "source": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
   },
   {
    "token": "anthropic-ai",
    "operator": "Anthropic",
    "purpose": "legacy",
    "source": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
   },
   {
    "token": "Claude-Web",
    "operator": "Anthropic",
    "purpose": "legacy",
    "source": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
   },
   {
    "token": "PerplexityBot",
    "operator": "Perplexity",
    "purpose": "search",
    "source": "https://docs.perplexity.ai/guides/bots"
   },
   {
    "token": "Perplexity-User",
    "operator": "Perplexity",
    "purpose": "user",
    "source": "https://docs.perplexity.ai/guides/bots"
   },
   {
    "token": "Google-Extended",
    "operator": "Google",
    "purpose": "training",
    "source": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
   },
   {
    "token": "Applebot-Extended",
    "operator": "Apple",
    "purpose": "training",
    "source": "https://support.apple.com/en-us/119829"
   },
   {
    "token": "CCBot",
    "operator": "Common Crawl",
    "purpose": "training",
    "source": "https://commoncrawl.org/ccbot"
   },
   {
    "token": "Bytespider",
    "operator": "ByteDance",
    "purpose": "training",
    "source": null
   },
   {
    "token": "Meta-ExternalAgent",
    "operator": "Meta",
    "purpose": "training",
    "source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
   },
   {
    "token": "meta-externalfetcher",
    "operator": "Meta",
    "purpose": "user",
    "source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
   },
   {
    "token": "Amazonbot",
    "operator": "Amazon",
    "purpose": "training",
    "source": "https://developer.amazon.com/amazonbot"
   },
   {
    "token": "cohere-ai",
    "operator": "Cohere",
    "purpose": "legacy",
    "source": null
   },
   {
    "token": "DuckAssistBot",
    "operator": "DuckDuckGo",
    "purpose": "search",
    "source": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot"
   },
   {
    "token": "MistralAI-User",
    "operator": "Mistral",
    "purpose": "user",
    "source": "https://docs.mistral.ai/robots"
   },
   {
    "token": "Googlebot",
    "operator": "Google",
    "purpose": "control",
    "source": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
   },
   {
    "token": "Bingbot",
    "operator": "Microsoft",
    "purpose": "control",
    "source": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0"
   },
   {
    "token": "Applebot",
    "operator": "Apple",
    "purpose": "control",
    "source": "https://support.apple.com/en-us/119829"
   }
  ],
  "note": "Operators rename crawlers and change what they do. Each quarterly edition re-checks these pages and records the labels it used; a changed purpose is reported as a change, not silently relabelled. ByteDance publishes no crawler documentation we could find."
 },
 "dateModified": "2026-09-28"
}