{
 "question": "What characteristics of a search are associated with Google showing an AI Overview, how much of the industry gap do they explain, and what happens after activation (which sources are cited, how they relate to the organic top 10, whether they support the sentence they are attached to, and how stable the result is over two days)?",
 "population": "Main sample: 800 US English keywords from 32 seed topics in eight industries (per industry the 40 highest-volume suggestions plus 60 drawn at random with 30+ monthly searches) and 160 question-form keywords (20 per industry), searched once on 26 September 2026. It estimates AI Overview exposure among commercially relevant keyword inventories in these industries, not across all Google searches. Matched-forms sample: 96 of the 800 (12 per industry, drawn at random, seed 20260928, excluding question-form and 'near me' keywords), each searched on 28 September 2026 as written, as a natural question and as a long non-question form of the same topic (288 searches). Grounding pilot: 120 AI Overviews from the 26 September searches (15 per industry, seed 20260928), one cited sentence each.",
 "inclusion": "All 800 main keywords for activation models; all 643 AI Overviews (main and question samples) for citation counts and source types; the 638 with at least one citation and an organic top 10 for overlap; the 96 matched bases with all three forms collected; the 97 grounding items whose cited page could be fetched.",
 "exclusion": "Grounding: 23 of 120 sampled sources could not be fetched or held no readable text (video, social and Google pages mostly) and are reported as unavailable, not coded.",
 "calculations": [
  "Activation model: logistic regression of AI Overview (yes/no) on industry, DataForSEO intent, local pack on the results page, 'near me', word count and log10 monthly volume; standard errors clustered by the 32 seed topics; odds ratios, average marginal effects, joint Wald tests.",
  "Industry before and after adjustment: average predicted probability with every keyword assigned to each industry in turn, from an industry-only model and from the full model.",
  "Local pack after adjustment: average predicted probability with the local pack set on and off for every keyword.",
  "Interactions: local pack x industry on the six industries with at least 10 keywords with and without a local pack; word count x intent.",
  "Predictive ladder: area under the ROC curve from 8-fold cross-validation grouped by seed topic, for nested sets of predictors.",
  "Price or cost words and 'best or top' are reported descriptively: every price keyword showed an AI Overview, so they cannot enter a logistic model.",
  "Matched forms: share of each form with an AI Overview; exact McNemar tests on the 96 pairs; logistic model of form plus word count, SE clustered by base keyword.",
  "Stability: the 96 original keywords on 26 and 28 September; Jaccard similarity of cited URLs and domains where both days showed an AI Overview.",
  "Citations: all distinct cited links of each AI Overview, normalized; top-10 overlap compares each cited URL (or its registrable domain) with the organic top 10 of the same results page.",
  "Grounding: for each sampled sentence, the first cited page was fetched and the six passages sharing most words with the sentence (up to 4,500 characters) were given to two independent model coders (Claude Sonnet and Claude Opus) who labeled supported, partial, not supported or unclear; agreement and Cohen's kappa reported; headline figures use items on which both agreed.",
  "Intervals: 95% bootstrap resampling the 32 seed topics (2,000 resamples, seed 20260926)."
 ],
 "limitations": [
  "The keyword sample leans to high-volume commercial searches in eight industries; the rate is a property of that sampling frame, not of Google searches in general.",
  "Observational: associations, not causes. The local pack is itself chosen by Google, so its association with fewer AI Overviews is not evidence that one suppresses the other.",
  "Matched forms were written by the research team (with an AI model) to keep the topic; they also add specificity, so question form and length are not fully separable.",
  "Stability covers two dates and 96 keywords only.",
  "Grounding is a model-coded pilot on 97 sentences, judged against automatically selected passages rather than the full page; no human coding.",
  "Intent labels come from DataForSEO's classifier; source types use a fixed domain list and do not separate brand-owned from editorial sites."
 ],
 "update": "Study 24 (week-over-week volatility, on or after 3 October 2026) will add a further date for these keywords; monthly after that.",
 "@context": "https://schema.org",
 "@type": "Dataset",
 "name": "When does Google show an AI Overview? 1,248 US searches",
 "version": "1.1",
 "dateCreated": "2026-09-26",
 "creator": {
  "@type": "Organization",
  "name": "Underneath",
  "url": "https://underneath.agency"
 },
 "url": "https://underneath.agency/research/ai-overviews-frequency-study",
 "temporalCoverage": "2026-09-26",
 "isAccessibleForFree": true,
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "code": "cite/pipeline/ (collection, validation and analysis scripts)",
 "variableMeasured": [
  "reanalyzed",
  "collected",
  "matched_collected",
  "n_main",
  "n_questions",
  "n_seeds",
  "overall",
  "questions",
  "volume_weighted",
  "by_industry",
  "by_intent",
  "by_local_pack",
  "by_near_me",
  "by_stratum",
  "composition_by_industry",
  "model",
  "local_pack_by_industry",
  "predictive_auc",
  "lexical",
  "by_volume_quintile",
  "denominator",
  "standardized",
  "sampling_frames",
  "question_within_seed",
  "matched",
  "stability",
  "citations",
  "source_types_pct",
  "source_types_by_intent_pct",
  "source_types_questions_pct",
  "overlap",
  "top_domains",
  "grounding",
  "derived"
 ],
 "dateModified": "2026-09-28",
 "research_questions": {
  "RQ1": "Activation: which query, intent and results-page characteristics are associated with an AI Overview, and does industry matter once they are held constant?",
  "RQ2": "Query form: when the same topic is searched as a keyword, a question and a long non-question form, how much does activation change?",
  "RQ3": "Denominator and sampling frame: how much does the headline rate depend on what is counted?",
  "RQ4": "Source selection: how many sources are cited, of which types, and how many rank in the organic top 10 of the same search?",
  "RQ5": "Grounding: do cited pages support the sentence they are attached to? (model-coded pilot)",
  "RQ6": "Stability: does the same keyword get the same outcome two days later?"
 },
 "hypotheses": {
  "H1": "Question-form searches are more likely to show an AI Overview after holding topic constant (matched forms).",
  "H2": "Longer searches are more likely to show one after adjustment.",
  "H3": "Searches whose results page shows a local pack are less likely to show one after adjustment for industry, intent, length and volume.",
  "H4": "Activation differs by DataForSEO intent after adjustment.",
  "H5": "Part of the industry gap is explained by the industries' query mix.",
  "H6": "AI Overview citations overlap only partly with the organic top 10.",
  "H7": "Source types differ by intent.",
  "H8": "A citation does not always mean the cited page supports the sentence."
 },
 "framework": "Visibility in AI search as stages: activation (is there an AI Overview), source selection (which pages are cited), grounding (does the cited page support the sentence), stability. Retrieval before citation, influence on the answer and user outcomes are named but not measured.",
 "comparison_sources": [
  {
   "source": "Xu, Iqbal and Montgomery, arXiv 2605.14021 (2026)",
   "url": "https://arxiv.org/abs/2605.14021",
   "figures": {
    "queries": 55393,
    "days": 40,
    "overall_activation_pct": 13.7,
    "question_form_activation_pct": 64.7,
    "cited_domains_not_on_first_page": "nearly 30%",
    "unsupported_claims_pct": 11.0,
    "atomic_claims": 98020,
    "categories": 19
   }
  }
 ]
}