{
 "question": "How faithfully do ChatGPT, Gemini, Perplexity and Google AI Mode turn a product’s published pricing into user-facing price claims, and where is fidelity lost: in the number, the plan, the billing condition, the source, or the coverage of plans?",
 "population": "Well-known software and subscription products with a public US pricing page, fetched on 26 September 2026.",
 "inclusion": "Products whose official pricing page could be fetched and showed at least three dollar amounts (other than free plans).",
 "exclusion": "Products whose pricing page failed to load or showed no prices.",
 "calculations": [
  "1.0 rule kept: each dollar amount above zero compared to the cent with the amounts on the captured page.",
  "1.1 coding: every amount coded for role (plan price, derived total, add-on, saving, other product, range, other) and, for plan prices, a verdict against the captured page (faithful, condition missing, wrong plan, plausible variant, differs, cannot be judged), the stated billing basis and the page price for the same plan.",
  "Coder: Claude Opus via the Claude Code CLI, one call per product with the four answers shuffled and the engine hidden; second coder Claude Sonnet on all 45 products, blind to the first; agreement reported as percent agreement and Cohen’s kappa. Model-coded; no person coded the data.",
  "Severity: relative difference between the quoted price and the page price for the same plan, for prices coded as differing.",
  "Source ladder for amounts not on the captured page: B = the amount appears on another vendor-owned page cited by any engine for the product; D = on a third-party page cited for the product; E = on no fetched source. Cited pages fetched over plain HTTP on 28 September 2026.",
  "Pricing complexity: count of six page features (billing toggle, per-seat pricing, usage or volume tiers, contact-sales plan, intro or promotional price, several products on one page), model-coded from the captured page.",
  "Uncertainty: 95% percentile intervals from 2,000 resamples of products; GEE logistic models of plan-price outcomes on engine, complexity and whether the answer cites a vendor-owned page, exchangeable correlation within product, with Wald tests."
 ],
 "limitations": [
  "The captured pricing page is the reference. It shows one billing toggle position, region and promotion; a price coded as a plausible variant or not supported may still be real. No price was confirmed with a vendor.",
  "All coding is by AI models (two Claude models), not people; agreement between them is reported, not accuracy against a human standard.",
  "One run per engine per product on 26 September 2026: run-to-run stability, prompt wording and staleness after a price change are not measured.",
  "Cited pages were fetched two days after the answers over plain HTTP; prices loaded by scripts or pages that block fetching are missed, so source-ladder shares are lower bounds.",
  "Plan coverage counts the paid plans on the captured page, including team, student and add-on editions where the page lists them, so full coverage is a demanding standard.",
  "45 products with a usable page; 15 of 60 were excluded. Perplexity queried through its API (sonar); ChatGPT and Gemini consumer apps and Google AI Mode through DataForSEO."
 ],
 "update": "quarterly",
 "@context": "https://schema.org",
 "@type": "Dataset",
 "name": "How faithfully do AI assistants quote software prices?",
 "version": "1.1",
 "dateCreated": "2026-09-26",
 "creator": {
  "@type": "Organization",
  "name": "Underneath",
  "url": "https://underneath.agency"
 },
 "url": "https://underneath.agency/research/ai-pricing-accuracy-study",
 "temporalCoverage": "2026-09-26",
 "isAccessibleForFree": true,
 "license": "https://creativecommons.org/licenses/by/4.0/",
 "code": "cite/pipeline/s15_analyze.py (1.0 rule), cite/pipeline/s15_v11.py (1.1 prep, coding, fetch, analysis, package)",
 "variableMeasured": [
  "price-claim role and verdict",
  "fidelity ladder",
  "billing basis stated",
  "severity and direction",
  "plan coverage",
  "source ladder",
  "citation support",
  "pricing complexity",
  "coder agreement"
 ],
 "dateModified": "2026-09-28",
 "framework": "A quoted price is treated as a claim with several dimensions (amount, plan, billing basis, unit, conditions, source), not as a string to find. Each claim is placed on a fidelity ladder: 0 not a plan price; 1 not supported by the captured page (differs, wrong plan, cannot be judged); 2 plausible variant not on the captured page; 3 right amount with a missing condition; 4 fully faithful. Answers are also scored for plan coverage, billing-basis disclosure and caveats. Temporal staleness, run-to-run stability and prompt sensitivity are named but not measured in this version.",
 "unit_of_analysis": "Price claim nested in answer, engine and product (965 dollar amounts, 180 answers, 45 products). Uncertainty and models are clustered by product.",
 "not_measured_next": {
  "phase_2_repeatability_and_prompts": "15 to 20 products x 4 engines x 5 runs x up to 6 controlled wordings (direct, explicit monthly, billing-aware, cite the official page, as of today, buyer framing); needs new collection spend.",
  "phase_3_staleness": "10 to 15 products whose pricing pages are captured weekly; after a detected price change, engines queried at 1, 3, 7, 14 and 30 days to measure how long the old price persists; needs new collection spend."
 },
 "hypotheses_tested": {
  "H1 engines differ": "joint engine Wald test in models",
  "H2 complexity lowers fidelity": "complexity coefficient",
  "H4 citing a vendor page raises fidelity": "cites_official coefficient",
  "H7 many page mismatches are context or variant, not fabrication": "verdict mix of amounts not on the page",
  "H8 engine x complexity": "interaction Wald test"
 }
}