{
 "question": "What searches do AI assistants write before answering a buyer question, and are those searches associated with which sources the answer cites?",
 "population": "80 US buyer questions (10 in each of eight industries) x 3 engines (ChatGPT GPT-5.4 nano and Gemini 3.5 Flash-Lite via DataForSEO LLM Responses API with web search; Claude Haiku 4.5 from study 7), one run each, 26 September 2026: 240 answers, 509 reported searches. Version 1.1 re-analyzes the same answers; no new engine queries.",
 "inclusion": "Every answer, including answers that reported no search (kept in denominators).",
 "exclusion": "",
 "calculations": [
  "Fixed phrase rules per search (year, review words, Reddit, site: operator, comparison, price, complaint, best/top, named source list), as in version 1.0.",
  "Function coding (Claude Opus via claude -p, blind to engine, order shuffled with seed 20260928): every function a search performs from a 12-function list, its primary function, and the role of any year (current, range with current, past edition of an annual ranking, past year needed by the question, past year unexplained).",
  "Second coder (Claude Sonnet), same blind items: agreement as percentage and Cohen's kappa.",
  "Reformulation distance: cosine similarity between each search and its question (sentence-transformers all-mpnet-base-v2).",
  "Named source -> citation: a fixed map of source names to registrable domains; for each answer and listed domain, whether the domain was named in a search and whether it was cited; compared with answers from the same engine and industry that did not name it; GEE logistic regression clustered by question.",
  "Fan-out breadth -> cited breadth: distinct cited registrable domains per answer against number of searches; Spearman within engine and GEE Poisson clustered by question with engine and industry fixed effects.",
  "Intervals: 95% bootstrap resampling questions (2,000 resamples, seed 20260926); all three engines' answers to a question move together."
 ],
 "limitations": [
  "One run per question and engine on one date: the stability of the searches, and of their link to citations, is not measured.",
  "The APIs report the searches but not the results each search returned, so retrieval and reranking are not observed; the link from searches to citations is at the answer level.",
  "Associations, not effects: engines, questions and searches are not randomized, so naming a source in a search is not shown to cause its citation.",
  "API models, not the consumer apps; the consumer ChatGPT and Gemini may search differently.",
  "Function and year-role labels are model-coded by two Claude models; no person coded them.",
  "Claims in the answers were not checked against the cited pages in this study."
 ],
 "update": "quarterly",
 "@context": "https://schema.org",
 "@type": "Dataset",
 "name": "The hidden searches AI assistants run before they answer",
 "version": "1.1",
 "dateCreated": "2026-09-26",
 "creator": {
  "@type": "Organization",
  "name": "Underneath",
  "url": "https://underneath.agency"
 },
 "url": "https://underneath.agency/research/ai-hidden-searches-study",
 "temporalCoverage": "2026-09-26",
 "isAccessibleForFree": true,
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "code": "cite/pipeline/ (collection, validation and analysis scripts)",
 "variableMeasured": [
  "reanalyzed",
  "collected",
  "answers",
  "questions",
  "searches",
  "bootstrap",
  "by_engine",
  "all_search_answers_have_citation",
  "answers_without_search",
  "answers_without_search_with_citation",
  "answers_with_search_without_citation",
  "identical_to_question",
  "similarity_model",
  "past_year_searches",
  "past_unexplained_searches",
  "breadth",
  "named_to_cited",
  "cited_domains_total",
  "cited_domains_named_in_search",
  "cited_domains_named_in_search_pct",
  "reddit",
  "function_link",
  "agreement",
  "regimes"
 ],
 "dateModified": "2026-09-28",
 "framework": "AI-search visibility as a pipeline: question -> fan-out searches -> retrieved results -> selected sources -> generated answer -> citation. This study measures the fan-out stage and its answer-level association with citation; retrieval, use and attribution of individual pages are not observed.",
 "research_questions": {
  "RQ1": "How many searches does each engine run, and how far do they move from the question?",
  "RQ2": "What functions do the searches perform (discovery, freshness, authority, reputation, price, comparison, platform targeting...)?",
  "RQ3": "Is fan-out breadth associated with the number of distinct domains cited, net of engine and industry?",
  "RQ4": "When a search names a source, is that source cited in the same answer more often than when it is not named?",
  "RQ5": "Are the searches and their effects stable across runs, dates and wordings? Not answered: one run per question and engine."
 }
}