{
 "question": "Of the \u201cbest X\u201d lists cited by AI search surfaces, how many rank their own publisher first; how does that relate to what conventional search shows; do answers mention the list\u2019s top pick; and how stable is it?",
 "population": "Pages cited by Google AI Overviews and AI Mode (800 US head and tail keywords) and by ChatGPT, Gemini, Perplexity and Claude (80 US buyer questions), collected 26 September 2026, whose title or address marks them as a ranked list; platform pages excluded. Secondary sets: repeat runs, prompt variants, four countries.",
 "interfaces": {
  "AI Overviews": "Google results page capture",
  "AI Mode": "Google results page capture",
  "ChatGPT": "consumer app capture (DataForSEO LLM Scraper)",
  "Gemini": "consumer app capture (DataForSEO LLM Scraper)",
  "Perplexity": "API, model sonar",
  "Claude": "API, model claude-haiku-4-5 with web search"
 },
 "instrument": "Numbered entries read with the longest run of consecutive numbered H2/H3 headings; publisher names from the web address plus the site name the page gives (og:site_name, Organization or publisher name, title suffix). Chosen on half of a 299-page gold standard and tested on the other half.",
 "validation": "Gold standard coded independently by two AI models (Claude Opus and Claude Sonnet), blind to the detector; disagreements adjudicated and logged. Model-coded, not human-coded.",
 "statistics": "95% intervals: Wilson (Clopper\u2013Pearson at 0 events) and bootstrap resampling publishers (lists) or questions (citations, answers), 2,000 resamples. Surface differences: logistic model with publisher-clustered errors.",
 "not_preregistered": true,
 "limitations": [
  "Self-first detection finds about two in three real cases (test sensitivity 64.9%), so shares are undercounts.",
  "368 of 1,574 list pages could not be fetched.",
  "Conventional-search top 10 is a proxy for what engines could have used, not their retrieval.",
  "Mentions are an observable proxy for representation, not evidence of influence.",
  "One collection date; repeat runs were minutes apart."
 ],
 "update": "A second collection date (repeat runs and reworded questions) is planned for October 2026.",
 "@context": "https://schema.org",
 "@type": "Dataset",
 "name": "How many \u201cbest of\u201d lists cited by AI rank their own brand first?",
 "dateCreated": "2026-09-26",
 "dateModified": "2026-09-27",
 "creator": {
  "@type": "Organization",
  "name": "Underneath",
  "url": "https://underneath.agency"
 },
 "url": "https://underneath.agency/research/self-promoting-best-lists-study",
 "isAccessibleForFree": true,
 "license": "https://creativecommons.org/licenses/by/4.0/"
}