[
  {
    "finding": "Half of the content that AI answer engines cite is less than 13 weeks old, according to Profound's Citation Decay data.",
    "description": "Profound's Citation Decay feature tracks week-over-week citations per URL from first-cited date (rise, peak, half-life, last-cited). The product post states half of cited content is under 13 weeks old but discloses no sample size, time window or per-platform breakdown.",
    "date": "2026-08-13",
    "application_area": "Content Freshness & Recency",
    "source": "Profound: Introducing Citation Decay in Profound",
    "source_link": "https://www.tryprofound.com/blog/citation-decay",
    "credibility": "Medium",
    "credibility_rationale": "Profound is a major answer-engine data vendor and the post is by technical staff (Joey Adelman, Matthew Huo), but it is a product announcement with no sample size, method or platform breakdown for the 13-week figure; widely re-cited by GEO blogs but not independently verified. Spot-check confirmed the sentence verbatim.",
    "evidence_type": "analysis",
    "publisher": "Profound",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Median age of AI-cited pages is 5.1 months on Claude, 6.0 on Google AI, 7.8 on Gemini and 8.0 on ChatGPT; 60% of Claude citations are under six months old vs 40% for ChatGPT.",
    "description": "Discovered Labs analyzed ~2 million prompt/engine/citation observations (through August 2026) across 10,000 crawled cited URLs for B2B SaaS prompts on ChatGPT, Claude, Google AI and Gemini. Page age had only a small standardized effect (+0.05) once other features were controlled.",
    "date": "2026-08",
    "application_area": "Content Freshness & Recency",
    "source": "Discovered Labs Research: What actually drives AI citations: a statistical analysis of 2M AI citations across 10K pages",
    "source_link": "https://discoveredlabs.com/research/what-drives-ai-citations",
    "credibility": "Medium",
    "credibility_rationale": "B2B AEO agency research (Liam Dunne, Ben Moore) with a commercial interest, but unusually transparent methodology (~2M tuples, 10K URLs, standardized regression with 95% CIs, nine robustness checks incl. Double ML and BH-FDR); cited by 5WPR's State of AI Citations 2026. Sample limited to B2B SaaS. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Discovered Labs",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "75% of pages cited by ChatGPT, Gemini and Perplexity were updated within the last year, 88% within two years, and only 3% of citations pointed to content over five years old.",
    "description": "Seer Interactive analyzed 7,683 dated cited pages with 47,097 citations across four industries, March-June 2026. Among 4,124 pages with both dates, 72% looked fresh by last-update but only 42% by original publish date, indicating updates rather than new publishing drive freshness. Within one year: Gemini 78%, ChatGPT 73%, Perplexity 65%.",
    "date": "2026-07-24",
    "application_area": "Content Freshness & Recency",
    "source": "Seer Interactive: Study: Content Recency's Impact on AI Visibility in 2026",
    "source_link": "https://www.seerinteractive.com/insights/study-content-recencys-impact-on-ai-visibility-in-2026",
    "credibility": "Medium",
    "credibility_rationale": "Established agency with documented method, but four verticals from its own client set and independent coverage unverified.",
    "evidence_type": "study",
    "publisher": "Seer Interactive (Sonny Vasquez)",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "In ChatGPT, pages 30-89 days old had the highest citation rate (32.8%) while pages under 30 days old underperformed (25.3%); 500-2,000 words was the optimal length band.",
    "description": "Same AirOps/Indig dataset. Pages over 5,000 words were cited 28.6%, under 500 words 30.5%; JSON-LD pages 38.5% vs 32.0% (correlational). Retrieval position dominated: 58.4% at position 1 vs 14.2% at position 10.",
    "date": "2026-04-13",
    "application_area": "Content Freshness & Recency",
    "source": "AirOps x Kevin Indig (Growth Memo): The Fan-Out Effect: What Happens Between a Query and a Citation",
    "source_link": "https://www.airops.com/report/the-fan-out-effect-what-happens-between-a-query-and-a-citation",
    "credibility": "Medium",
    "credibility_rationale": "Same vendor-published correlational study with large described sample; spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "AirOps and Kevin Indig (Growth Memo)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Each year of page age cuts AI search visibility roughly 40-60% (42-43% per year), with Perplexity the most recency-biased (1.69x) and Gemini the least (0.78x).",
    "description": "Gander analyzed 194,077 URLs in Q1 2026 (vs a 14,764-URL Dec 2025 baseline) across ChatGPT, Google AIO, Gemini and Perplexity at retrieval and citation layers. Retrieval-layer 2026:2025 ratios: Perplexity 1.69x, AIO 1.06x, Gemini 0.78x, ChatGPT 0.28x; two-year-old content sits at ~33% of peak; 23.4% of ChatGPT fan-out queries injected a year.",
    "date": "2026-04-07",
    "application_area": "Content Freshness & Recency",
    "source": "Gander: The 1-Year Half-Life: How Content Freshness Drives Visibility in AI Search",
    "source_link": "https://takeagander.ai/resources/gander-blog/how-content-freshness-drives-visibility-in-ai-search/",
    "credibility": "Medium",
    "credibility_rationale": "Smaller vendor but methodology-documented with an appendix; vendor-published, no peer review, no corroboration verified.",
    "evidence_type": "analysis",
    "publisher": "Gander (Mehrad Soltani)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Microsoft Bing says that for AI-powered search, 'freshness signals directly influence how quickly updates are reflected in search results and AI generated answers,' and recommends accurate sitemap lastmod plus IndexNow.",
    "description": "Bing Webmaster Blog (July 31, 2025, Fabrice Canel and Krishna Madhavan): sitemap lastmod 'remains a key signal' for recrawl prioritization, Bing typically revisits sitemaps at least once per day, and accurate lastmod matters as AI search adjusts surfacing in near real time; combine sitemaps with IndexNow for real-time URL-level submission so content stays visible in Copilot and Bing AI answers.",
    "date": "2025-07-31",
    "application_area": "Content Freshness & Recency",
    "source": "Bing Webmaster Blog: Keeping Content Discoverable with Sitemaps in AI Powered Search",
    "source_link": "https://blogs.bing.com/webmaster/July-2025/Keeping-Content-Discoverable-with-Sitemaps-in-AI-Powered-Search",
    "credibility": "High",
    "credibility_rationale": "Official Microsoft Bing Webmaster Blog post by Bing Principal Product Managers; quoted sentences verified verbatim. (Merged from two duplicate rows.)",
    "evidence_type": "official-guidance",
    "publisher": "Microsoft Bing",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "AI assistants cite content 25.7% fresher than organic Google results (1,064 vs 1,432 days old on average), with ChatGPT citing URLs 458 days newer than organic results.",
    "description": "Ahrefs analyzed 16.975M cited URLs across ChatGPT, Perplexity, Gemini, Copilot, Google AIO and organic SERPs. Average age: ChatGPT 958 days, ChatGPT in-text 1,023, Copilot 1,056, Gemini 1,118, Perplexity 1,166, AI Overviews 1,432 (matching organic). AI-cited pages still average ~2.9 years old vs 3.9 for organic.",
    "date": "2025-07-28",
    "application_area": "Content Freshness & Recency",
    "source": "Ahrefs: New Study: AI Assistants Prefer to Cite 'Fresher' Content (17 Million Citations Analyzed)",
    "source_link": "https://ahrefs.com/blog/do-ai-assistants-prefer-to-cite-fresh-content/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor; sample, per-platform averages and the (1432-1064)/1432 calculation shown; Ryan Law with data by Xibeijia Guan. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "The median half-life of an AI citation is about 4-5 weeks: ChatGPT ~3.4 weeks, Google AI surfaces 4.3-4.8 weeks, Perplexity ~5.8 weeks.",
    "description": "Scrunch and Stacker analyzed 3.5M citation events (Sept 2025-Mar 2026) across ChatGPT, Google AIO/AI Mode/Gemini and Perplexity, measuring how quickly a source's citation activity halves (4.5 weeks overall). Stacker Partner Network editorial domains showed ~2x longevity (up to ~12.3 weeks on Perplexity). Page undated.",
    "date": "unknown",
    "application_area": "Content Freshness & Recency",
    "source": "Scrunch x Stacker: The Half-Life of AI Citations: What 3.5 Million Citation Events Taught Us About AI's Memory",
    "source_link": "https://scrunch.com/blog/half-life-of-ai-citations",
    "credibility": "Medium",
    "credibility_rationale": "Monitoring vendor plus syndication network with a large sample and described method, but vendor-published, undated, and the partner comparison promotes Stacker's product. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "Scrunch AI with Stacker (Michael Iannelli)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "In ChatGPT, 52% of cited listicles were cited 2+ times, while Google AI Mode favored official brand product/category pages and 64% of Perplexity-surfaced URLs never got cited.",
    "description": "Peec AI analyzed 1M+ citations across ChatGPT, AI Mode and Perplexity to set citation-rate benchmarks (ChatGPT 2.0+, AI Mode 1.1-1.5, Perplexity 1.5-2.0); ~10-11% of URLs produced 37% of citations. Research conducted Feb 27, 2026 (Tom Wells), published Sept 18, 2026.",
    "date": "2026-09-18",
    "application_area": "Content Format & Structure",
    "source": "Peec AI: What Does a Good Citation Rate Look Like? Benchmarks From Over 1 Million AI Citations",
    "source_link": "https://peec.ai/blog/citation-rate-benckmarks-from-over-1-million-citations",
    "credibility": "Medium",
    "credibility_rationale": "Venture-backed vendor with a large sample and stated benchmarks, but vendor-published and uncorroborated. Spot-check confirmed page reachability and figures.",
    "evidence_type": "study",
    "publisher": "Peec AI",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Prompt-content alignment is the strongest page-level AI citation driver (standardized effect +0.37, about 3x the next signal), while domain authority is the strongest overall (SHAP 0.38).",
    "description": "Same Discovered Labs regression on ~2M observations and 10,000 pages: alignment +0.37 (95% CI +0.33 to +0.41, ~30% more citations per SD); page length +0.13, title-prompt similarity +0.09, FAQ sections +0.07, TLDR blocks +0.05, page age +0.05, author bios +0.02.",
    "date": "2026-08",
    "application_area": "Content Format & Structure",
    "source": "Discovered Labs Research: What actually drives AI citations: a statistical analysis of 2M AI citations across 10K pages",
    "source_link": "https://discoveredlabs.com/research/what-drives-ai-citations",
    "credibility": "Medium",
    "credibility_rationale": "Agency-published with commercial interest but transparent method (standardized regression, CIs, nine robustness checks) and cited by 5WPR; sample limited to B2B SaaS prompts. Spot-check confirmed the +0.37 effect and CI.",
    "evidence_type": "study",
    "publisher": "Discovered Labs",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Google says structured data, llms.txt, AI text files, Markdown, special markup and content 'chunking' are not required to appear in its generative AI search features.",
    "description": "Google's 'Guide to Optimizing for Generative AI Features on Google Search' (updated 2026-07-10): structured data isn't required and there's no special schema; no new machine-readable files, AI text files, markup or Markdown are needed; no requirement to break content into tiny pieces; no need to write in a specific way for AI.",
    "date": "2026-07-10",
    "application_area": "Content Format & Structure",
    "source": "Google Search Central: Google's Guide to Optimizing for Generative AI Features on Google Search",
    "source_link": "https://developers.google.com/search/docs/fundamentals/ai-optimization-guide",
    "credibility": "High",
    "credibility_rationale": "Official Google Search Central documentation; covered by Search Engine Journal and Search Engine Land.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Adding JSON-LD schema to 1,885 pages produced no meaningful AI citation lift: -4.6% on Google AI Overviews, +2.4% on AI Mode and +2.2% on ChatGPT (the latter two indistinguishable from zero).",
    "description": "Ahrefs tracked 1,885 pages that added JSON-LD (Aug 2025-Mar 2026) against 4,000 matched controls using 30-day windows and four tests (t-test, matched DiD, event study, symmetrical-window DiD). The AIO decline was significant but too small to attribute to schema; 53% of AI-cited pages use schema.",
    "date": "2026-05-11",
    "application_area": "Content Format & Structure",
    "source": "Ahrefs: We Tracked 1,885 Pages Adding Schema. AI Citations Barely Moved.",
    "source_link": "https://ahrefs.com/blog/schema-ai-citations/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor; matched-control before/after design with four documented tests; covered by Search Engine Journal. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "ChatGPT cites fewer sources per prompt (6.88 vs 12.06 Google, 16.35 Perplexity) but its cited pages have ~4x higher citation influence; longer, structured, evidence-rich pages absorb more.",
    "description": "Zhang, He and Yao separate citation selection from citation absorption, computing an influence score per citation on the public geo-citation-lab dataset (602 prompts, 21,143 citations, 18,151 pages, 72 features). Average influence 0.2713 ChatGPT vs 0.0584 Google and 0.0646 Perplexity; high-influence pages were longer, structured and rich in extractable evidence. Preprint.",
    "date": "2026-04-28",
    "application_area": "Content Format & Structure",
    "source": "arXiv: From Citation Selection to Citation Absorption: A Measurement Framework for Generative Engine Optimization Across AI Search Platforms (Zhang, He, Yao)",
    "source_link": "https://arxiv.org/abs/2604.25707",
    "credibility": "Medium",
    "credibility_rationale": "Independent researchers with no institutional affiliation, not peer-reviewed, corroboration only in aggregators and a vendor blog; mitigated by public dataset/code and a documented method.",
    "evidence_type": "paper",
    "publisher": "Zhang Kai, He Xinyue, Yao Jingang (independent researchers)",
    "publisher_type": "independent-expert"
  },
  {
    "finding": "Pages whose headings closely match the query (0.90+ similarity) were cited by ChatGPT 41.0% of the time vs 30.2% for weak matches, the strongest on-page lever measured.",
    "description": "AirOps and Kevin Indig analyzed 16,851 queries, 353,799 pages and 50,553 ChatGPT responses. Heading-query similarity beat word count and topical breadth; pages covering 26-50% of fan-out sub-queries were cited 38.2% vs 34.0% for 100% coverage. Correlational.",
    "date": "2026-04-13",
    "application_area": "Content Format & Structure",
    "source": "AirOps x Kevin Indig (Growth Memo): The Fan-Out Effect: What Happens Between a Query and a Citation",
    "source_link": "https://www.airops.com/report/the-fan-out-effect-what-happens-between-a-query-and-a-citation",
    "credibility": "Medium",
    "credibility_rationale": "Large described sample co-authored with a recognized SEO researcher, but an AI content-platform vendor, correlational, and independent coverage unverified. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "AirOps and Kevin Indig (Growth Memo)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Listicles earned 21.9% of all AI citations (articles 16.7%, product pages 13.7%) and 40% of commercial-intent citations; query intent predicted cited content type better than industry or model.",
    "description": "Wix Studio AI Search Lab analyzed 75,000 AI answers and 1,056,727 citations from ChatGPT, Google AI Mode and Perplexity (data via the Peec AI platform). Articles were cited 2.7x more for informational queries; Perplexity uniquely favored discussion pages (17%).",
    "date": "2026-03-23",
    "application_area": "Content Format & Structure",
    "source": "Wix Studio AI Search Lab: The content types most cited by LLMs",
    "source_link": "https://www.wix.com/studio/ai-search-lab/research/content-types-most-cited-by-llms",
    "credibility": "High",
    "credibility_rationale": "Major publicly traded web platform; large sample with content-type and intent breakdowns. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Wix Studio AI Search Lab",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Even when cited, short sources were under-used in AI answers by 13.4 pp (SearchGPT) and 17.7 pp (Google AIO), and AIO under-used cited social/forum sources by 22.1 pp.",
    "description": "The 'Answer Bubbles' study measured source-summary fidelity, how much each cited source contributes to the generated answer. Long sources gained +3.9 pp (SearchGPT) and +4.6 pp (AIO) in representation, Google AI Overviews under-represented negatively framed sources by 13.8 pp, and search grounding reduced hedging language by 40-60% (certainty-to-tentative ratio rose from 0.39 to 0.49), so being cited is not the same as being absorbed into the answer.",
    "date": "2026-03-17",
    "application_area": "Content Format & Structure",
    "source": "arXiv / EMNLP 2026: Answer Bubbles: Information Exposure in AI-Mediated Search (arXiv 2603.16138)",
    "source_link": "https://arxiv.org/html/2603.16138v2",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (arXiv comments state EMNLP 2026). Authors are at UIUC's Siebel School of Computing and Data Science; Chandrasekharan and Saha are established computational social science researchers. Large sample (11,000 real queries across five systems) with documented methodology for source-summary fidelity.",
    "evidence_type": "paper",
    "publisher": "University of Illinois Urbana-Champaign (Siebel School of Computing and Data Science) via arXiv; accepted to EMNLP 2026",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Optimizing entity density lifted citation probability 292% for mid-tail queries; cited URLs average 1,800 words vs 1,200, but adding fluff without entities lowers citations.",
    "description": "iPullRank (Francine Monahan) analyzed 79,000+ URL-query pairs across ChatGPT, Claude, Perplexity and Google. Raising word count without entity count decreased citation probability; ranking position remained the main gatekeeper. Authors flag findings as correlational bucket-level aggregations.",
    "date": "2026-03-12",
    "application_area": "Content Format & Structure",
    "source": "iPullRank: Beyond Rankings: Designing AI Search Metrics for the Next Era of SEO",
    "source_link": "https://ipullrank.com/ai-search-metrics",
    "credibility": "Medium",
    "credibility_rationale": "Well-known technical SEO agency (Mike King, SEL AI Search Marketer of the Year 2025) with stated sample and explicit caveats, but agency analysis with limited raw method detail and no external corroboration of the 292% figure.",
    "evidence_type": "analysis",
    "publisher": "iPullRank",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "LLM-cited pages outscored Google-ranking pages on clarity and summarization (+32.8%), EEAT signals (+30.6%), Q&A format (+25.5%), section structure (+22.9%) and structured data (+21.6%).",
    "description": "Semrush compared 304,805 URLs cited by ChatGPT Search, Google AI Mode and Perplexity (11,882 prompts) against 921,614 URLs in Google's top 20 (59,410 keywords), collected July 15-Aug 6, 2025, scoring 13 visible-text parameters. 'Non-promotional tone' was negative (-26.2%).",
    "date": "2026-01-14",
    "application_area": "Content Format & Structure",
    "source": "Semrush: How We Built a Content Optimization Tool for AI Search [Study]",
    "source_link": "https://semrush.com/blog/content-optimization-ai-search-study",
    "credibility": "High",
    "credibility_rationale": "Large publicly traded SEO data vendor with documented positive/negative samples and window. Spot-check confirmed all deltas and sample sizes.",
    "evidence_type": "study",
    "publisher": "Semrush",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Across 174,048 pages cited in Google AI Overviews, word count showed near-zero correlation with citations (Spearman 0.04) and 53.4% of citations went to pages under 1,000 words.",
    "description": "Ahrefs analyzed 560,346 AIOs and 1,677,876 cited URLs (174,048 with word-count data; average 1,282 words). Citation share: under 350 words 16.6%, 350-1,000 36.8%, 1,000-2,000 30.6%, over 2,000 16.0%.",
    "date": "2025-12-03",
    "application_area": "Content Format & Structure",
    "source": "Ahrefs: Short vs. Long Content in AI Overviews: The Data Says Both Work",
    "source_link": "https://ahrefs.com/blog/short-vs-long-content-in-ai-overviews/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor with documented sample, statistic and named author/data contributor/reviewer (Gavoyannis, Guan, Law).",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "In an e-commerce GEO testbed, most hand-crafted rewriting heuristics underperformed the baseline; a prompt meta-optimizer improved 63 of 75 prompt-engine cells and transferred in 13 of 15.",
    "description": "E-GEO (Bagga, Farias, Korkotashvili, Peng, Wu) pairs 13,747 multi-sentence Amazon product queries from r/BuyItForLife with 10 retrieved listings each and tests 15 rewriting heuristics, 7 LLM rewriters and 5 generative engines/re-rankers. With GPT-4.1 as rewriter, only 'trick' (+0.14 row-mean rank), 'FAQ' (+0.05), 'competitive' (+0.03) and 'format' (+0.02) matched or beat the baseline (-0.03) while 'advertisement' (-1.82), 'language' (-1.66), 'minimalist' (-1.49) and 'storytelling' (-4.36) hurt; prompts optimized on GPT-4.1 improved over their heuristic starting points in 13 of 15 cases on Claude, and a 'questionable content' clause cut the flag rate of red-teamed prompts by 60 points.",
    "date": "2025-11-25",
    "application_area": "Content Format & Structure",
    "source": "arXiv (Bagga, Farias, Korkotashvili, Peng, Wu; MIT/Columbia): E-GEO: A Testbed for Generative Engine Optimization in E-Commerce",
    "source_link": "https://arxiv.org/html/2511.20867",
    "credibility": "High",
    "credibility_rationale": "Authors are affiliated with MIT (Farias is a senior MIT Sloan operations professor) and Columbia Business School; large documented testbed (13,747 queries x 10 listings from a 17M-product corpus, 2,000 test queries, 15 heuristics x 7 rewriters x 5 engines) with per-cell results and defense ablations. Preprint with no listed peer-review venue.",
    "evidence_type": "paper",
    "publisher": "arXiv preprint; MIT Operations Research Center, MIT Sloan, MIT EECS, Columbia Business School",
    "publisher_type": "research-institution"
  },
  {
    "finding": "AutoGEO, an LLM rewriter that learns generative-engine preference rules, improved content visibility by up to 50.99% over the strongest baseline on GEO-Bench.",
    "description": "Wu, Zhong, Kim and Xiong (CMU, ICLR 2026) extracted engine preference rules (comprehensiveness, accuracy, structure, clarity, citations, depth) and used them for AutoGEO-API and as rewards for AutoGEO-Mini. On GEO-Bench, Researchy-GEO and an e-commerce set with three engines, AutoGEO-API gained 34.37%/42.87%/33.52% word-based visibility; AutoGEO-Mini +20.99% at ~0.0071x cost.",
    "date": "2025-10-13",
    "application_area": "Content Format & Structure",
    "source": "arXiv / ICLR 2026 (Carnegie Mellon University): What Generative Search Engines Like and How to Optimize Web Content Cooperatively (AutoGEO)",
    "source_link": "https://arxiv.org/abs/2510.11438",
    "credibility": "High",
    "credibility_rationale": "Carnegie Mellon authors; accepted at ICLR 2026 with public code; documented benchmarks and engines.",
    "evidence_type": "paper",
    "publisher": "Yujiang Wu, Shanshan Zhong, Yubin Kim, Chenyan Xiong (Carnegie Mellon University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "In a 1,702-citation audit across Brave, Google AI Overviews and Perplexity, Metadata & Freshness, Semantic HTML and Structured Data pillars were most associated with citation.",
    "description": "Kumar and Palkhouski collected 1,702 citations for 70 product-intent B2B SaaS prompts, auditing 1,100 URLs against the 16-pillar GEO-16 rubric with logistic models (domain-clustered SEs). Pages scoring >=0.70 with >=12 pillar hits had substantially higher citation rates. Observational, English-only.",
    "date": "2025-09-13",
    "application_area": "Content Format & Structure",
    "source": "arXiv (UC Berkeley / Wrodium Research): AI Answer Engine Citation Behavior: An Empirical Analysis of the GEO-16 Framework",
    "source_link": "https://arxiv.org/abs/2509.10762",
    "credibility": "Medium",
    "credibility_rationale": "Preprint with UC Berkeley and GEO-vendor (Wrodium) affiliations, not peer-reviewed, small B2B SaaS sample, authors acknowledge possible confounding; secondary coverage mostly vendor blogs.",
    "evidence_type": "paper",
    "publisher": "Arlen Kumar, Leanid Palkhouski (UC Berkeley / Wrodium Research)",
    "publisher_type": "independent-expert"
  },
  {
    "finding": "C-SEO Bench finds most conversational-SEO/GEO tactics ineffective or harmful to LLM-answer ranking, traditional SEO more effective, and gains shrink as more competitors adopt them.",
    "description": "Puerto et al. (Parameter Lab / NAVER AI Lab, NeurIPS 2025 D&B) built C-SEO Bench: two tasks, six domains, nine C-SEO methods, 1,921 queries, varying the number of adopters. Most methods were ineffective or reduced ranking, traditional SEO was significantly more effective, and gains decreased as adoption spread, contrasting with the original GEO paper's 40% lift.",
    "date": "2025-06-06",
    "application_area": "Content Format & Structure",
    "source": "arXiv / NeurIPS 2025 Datasets & Benchmarks (Parameter Lab, NAVER AI Lab): C-SEO Bench: Does Conversational SEO Work?",
    "source_link": "https://arxiv.org/abs/2506.11097",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (NeurIPS 2025 Datasets & Benchmarks Track) with open-source benchmark code.",
    "evidence_type": "paper",
    "publisher": "Haritz Puerto, Martin Gubri, Tommaso Green, Seong Joon Oh, Sangdoo Yun (Parameter Lab / NAVER AI Lab)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "StealthRank passages embedded in product descriptions improved target rank by up to 3.9 positions over Strategic Text Sequences (1.46 vs 5.40 on Mistral-7B) at far lower perplexity.",
    "description": "Tang, Fan, Yu, Yang, Zhao and Hu use energy-based optimization with Langevin dynamics to produce fluent adversarial passages that manipulate LLM rankers. Across Llama-3.1-8B-Instruct, Vicuna-7B-v1.5, Mistral-7B-Instruct-v0.3 and DeepSeek-LLM-7B-Chat, StealthRank kept perplexity at 51-109 versus 2,449-195,939 for STS, outperformed STS by 1.5-2.2 rank points on Ragroll, matched or surpassed Tree-of-Attacks on 3 of 4 models per dataset with bad-word ratios of 0.10-0.48 vs 0.50-0.76, and was rated more fluent, persuasive and less detectable across 183 human judgments.",
    "date": "2025-04-08",
    "application_area": "Content Format & Structure",
    "source": "arXiv (Tang, Fan, Yu, Yang, Zhao, Hu; USC/ASU): StealthRank: LLM Ranking Manipulation via Stealthy Prompt Optimization",
    "source_link": "https://arxiv.org/html/2504.05804v2",
    "credibility": "High",
    "credibility_rationale": "University research (USC, ASU) with a fully documented method, public code on GitHub, four open models tested on two datasets, tabulated rank/perplexity/bad-word-ratio results and a 183-judgment human study; used as a baseline in GEO-Bench and cited by subsequent ranking-manipulation papers. Preprint with no peer-review venue listed.",
    "evidence_type": "paper",
    "publisher": "arXiv preprint; University of Southern California and Arizona State University",
    "publisher_type": "research-institution"
  },
  {
    "finding": "In June 2024, only 8.71% of 100,013 keywords triggered AI Overviews; average AIO text was 4,342 characters and 84.72% of AIOs linked to at least one top-10 organic domain.",
    "description": "SE Ranking's early post-rollout study of 100,013 keywords across 20 niches (data collected June 3, 2024) found the 8.71% trigger rate was a 52.76% drop from the prior study and average AIO length rose 24.59% from 3,485 characters. The most common pre-click link count was 1 (average 2.2) and post-click 4 (average 5.5); featured snippets appeared alongside AIOs 45.39% of the time with matching sources in 61.79% of those cases, and ads accompanied AIOs 87% of the time, up from 73% pre-rollout.",
    "date": "2024-06-19",
    "application_area": "Content Format & Structure",
    "source": "SE Ranking: Google AI Overviews: New Research Study",
    "source_link": "https://seranking.com/blog/google-ai-overviews-research/",
    "credibility": "High",
    "credibility_rationale": "SE Ranking is an established SEO data platform; this large-sample study (100,013 keywords, 20 niches, collected June 3, 2024, author Yevheniia Khromova) documents its methodology and comparison baseline, and its trigger-rate and link-count figures were widely cited in 2024 coverage of the AIO rollout.",
    "evidence_type": "study",
    "publisher": "SE Ranking",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "An optimized Strategic Text Sequence on a product page took a never-recommended product to an LLM's top pick within 100 GCG iterations, improving its rank in ~40% of 200 trials.",
    "description": "Kumar and Lakkaraju (Harvard) built a catalog of fictitious coffee machines and used the Greedy Coordinate Gradient algorithm on Llama-2 to craft a Strategic Text Sequence appended to a target product page. For 'ColdBrew Master' ($199) a fixed-order STS improved rank in about 40% of 200 random-order evaluations with no change in about 60%, and optimizing over random orderings significantly increased the advantage with negligible downside; for 'QuickBrew Express' ($89), never top-ranked before, the STS significantly increased its chance of first position. Transfer to GPT-3.5/GPT-4 was not directly tested.",
    "date": "2024-04-11",
    "application_area": "Content Format & Structure",
    "source": "arXiv (Kumar & Lakkaraju, Harvard University): Manipulating Large Language Models to Increase Product Visibility",
    "source_link": "https://arxiv.org/html/2404.07981v2",
    "credibility": "High",
    "credibility_rationale": "Authored by Harvard University researchers (Lakkaraju runs a well-known trustworthy-AI lab) with a fully documented method (GCG optimization on Llama-2, fictitious catalog, 200 independent random-order evaluations). Preprint not marked peer-reviewed, but widely cited by later GEO and ranking-manipulation papers.",
    "evidence_type": "paper",
    "publisher": "arXiv preprint; Aounon Kumar and Himabindu Lakkaraju, Harvard University",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Adding quotations, statistics or source citations raised page visibility in generative-engine answers 30-40% (best method 41%), while keyword stuffing scored below baseline.",
    "description": "Aggarwal et al. (Princeton/IIT Delhi, KDD 2024) built GEO-bench (10K queries), rewrote source pages with nine tactics and measured Position-Adjusted Word Count and Subjective Impression in a GPT-3.5-turbo engine. Vs a 19.5 baseline: Quotation Addition 27.8, Statistics 25.9, Fluency 25.1, Cite Sources 24.9, Keyword Stuffing 17.8; on Perplexity.ai Quotation Addition +22%, Keyword Stuffing -10%.",
    "date": "2023-11-16",
    "application_area": "Content Format & Structure",
    "source": "arXiv / KDD 2024 (Princeton, IIT Delhi): GEO: Generative Engine Optimization",
    "source_link": "https://arxiv.org/abs/2311.09735",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed at ACM SIGKDD 2024 (DOI 10.1145/3637528.3671900), Princeton/IIT Delhi authors, fully documented benchmark and released code; widely cited by later papers and industry.",
    "evidence_type": "paper",
    "publisher": "Pranjal Aggarwal, Vishvak Murahari, Tanmay Rajpurohit, Ashwin Kalyan, Karthik Narasimhan, Ameet Deshpande (Princeton University / IIT Delhi); ACM KDD 2024",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Neural retrievers and re-rankers systematically rank LLM-generated documents above human-written ones ('source bias'), attributed to LLM text's more focused, lower-noise semantics.",
    "description": "Dai et al. (Renmin University, KDD 2024) evaluated IR models on mixed human/LLM corpora and found consistent preference for LLM-generated text from first-stage retrievers through re-rankers; text-compression analysis attributes it to lower-noise semantics. Relevant to how RAG answer engines select sources.",
    "date": "2023-10-31",
    "application_area": "Content Format & Structure",
    "source": "arXiv / ACM KDD 2024: Neural Retrievers are Biased Towards LLM-Generated Content (Dai et al.)",
    "source_link": "https://arxiv.org/abs/2310.20501",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (KDD 2024, DOI 10.1145/3637528.3671882); Renmin University authors; public code and datasets; widely cited in IR literature.",
    "evidence_type": "paper",
    "publisher": "Sunhao Dai et al. (Renmin University of China / CAS)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "AI Overviews appeared on 52.63% of 100,000 French searches two weeks after launch; 60.46% of cited pages ranked in the organic top 10 and YouTube appeared in 51.31% of AIOs.",
    "description": "SE Ranking analyzed 100,000 keywords across 20 niches in France, with data collected August 7-8, 2026, roughly two weeks after AIOs arrived in France on July 22. Across 52,564 AIOs with source data, answers averaged 6.91 citations (57.19% contained 5-8 sources), 76.33% of cited pages appeared in the organic top 50, YouTube accounted for 14.33% of citations, French-language sources made up 89.34% of classified sources, and Business (87.1%) and Relationships (86.04%) niches triggered AIOs most versus Fashion & Beauty at 20.28%.",
    "date": "2026-08-18",
    "application_area": "Citations & Source Selection",
    "source": "SE Ranking: AI Overviews Already Appear on Over 52% of French Searches",
    "source_link": "https://seranking.com/blog/ai-overviews-france-study/",
    "credibility": "High",
    "credibility_rationale": "SE Ranking is an established SEO data platform publishing a large-sample study (100,000 keywords, 20 niches, 52,564 AIOs, 363,330 citations from 30,383 domains) with documented methodology and collection dates (August 7-8, 2026); findings were independently reported by Siècle Digital alongside Ahrefs' own France AIO study.",
    "evidence_type": "study",
    "publisher": "SE Ranking",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Passages AI answers cited with attribution showed a visible 2025-2026 date 80% of the time vs 53% for uncited passages, and named the entity at first mention 96% vs 82%.",
    "description": "Advanced Web Ranking hand-coded 112 passages from 265 citations across 40 queries on Google AIO and Bing Copilot Search (single day, July 29, 2026). Cited passages were less often pure consensus (61% vs 82%); 59% of AIO citations in 12 fully captured answers pointed to Reddit, YouTube or LinkedIn. Small sample, coding sheet published.",
    "date": "2026-08-13",
    "application_area": "Citations & Source Selection",
    "source": "Advanced Web Ranking: What Gets Quoted and What Gets Absorbed: A Passage-Level Study of AI Citations",
    "source_link": "https://www.advancedwebranking.com/blog/passages-quoted-vs-passages-absorbed-in-ai-answers",
    "credibility": "Medium",
    "credibility_rationale": "Long-established rank-tracking vendor with an openly published method and coding sheet, but a very small single-day, single-coder sample.",
    "evidence_type": "study",
    "publisher": "Advanced Web Ranking (Bart Magera)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "News content is only 7.2% of AI citations across seven platforms, and publishers with OpenAI licensing deals earned 10.2 ChatGPT citations per page vs 6.9 unlicensed (48% premium).",
    "description": "OtterlyAI and Press Ranger analyzed 129.3M citations across seven platforms in June 2026 against 91 licensing deals covering 314 publisher domains. Claude cited news most (14.0%); 58.8% of licensed publishers' citations came from commercial evergreen pages vs 5.1% for dated news. The premium appeared only for OpenAI partners. No authors or sampling method given.",
    "date": "2026-08",
    "application_area": "Citations & Source Selection",
    "source": "OtterlyAI x Press Ranger: AI News Content Licensing Deals Study",
    "source_link": "https://insights.otterly.ai/ai-news-content-licensing-deals/",
    "credibility": "Medium",
    "credibility_rationale": "Established monitoring vendor with a large sample, but no named authors, no sampling/classification methodology and no independent coverage found.",
    "evidence_type": "study",
    "publisher": "OtterlyAI and Press Ranger",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "LLMs ground brand-related answers in third-party sources 85.7% of the time and the brand's own site only 14.3%; Wikipedia is the top-cited domain in 11 of 12 European languages.",
    "description": "arXiv preprint analyzing 167,551 URL-grounded citations for 128 brands across 12 markets and 13 languages using Perplexity Sonar Pro, Gemini 3.1 Pro and GPT-5.4. 80% of citations came from ~18% of domains (Zipf exponent 0.86); owned-site share: Perplexity 16.8%, GPT 12.9%, Gemini 5.8%.",
    "date": "2026-06-24",
    "application_area": "Citations & Source Selection",
    "source": "arXiv (Zatuchin, EUAS / Rankfor.AI): How Large Language Models Source Brand Reputation Across Languages and Markets",
    "source_link": "https://arxiv.org/html/2606.25787v1",
    "credibility": "Medium",
    "credibility_rationale": "Single-author preprint, not peer-reviewed, with a vendor (Rankfor.AI) conflict of interest; mitigated by fully documented methodology and an academic affiliation. No independent citations located.",
    "evidence_type": "paper",
    "publisher": "Dmitrij Zatuchin (Estonian Entrepreneurship University of Applied Sciences; Rankfor.AI)",
    "publisher_type": "independent-expert"
  },
  {
    "finding": "About 16% of unique sources cited by generative search engines show evidence of being AI-generated, from 27.8% for Copilot to 14.7% Gemini, 9.4% Perplexity and 7.3% ChatGPT.",
    "description": "Allaham and Diakopoulos (Northwestern) audited four engines on 712 real-world queries (politics 175, health 257, environment 280) with AI-text detection on cited sources. Health had the highest AI-generated share (16.3%); 59.1% of cited domains were cited only once. Preprint v1.",
    "date": "2026-05-22",
    "application_area": "Citations & Source Selection",
    "source": "arXiv (Northwestern University): Synthetic Sources?: Auditing Generative Search Engine Citations for Evidence of AI-Generated Sources (Allaham, Diakopoulos)",
    "source_link": "https://arxiv.org/abs/2605.23684",
    "credibility": "High",
    "credibility_rationale": "Northwestern authors (Nicholas Diakopoulos, established computational journalism professor); covered by Library Journal infoDOCKET; documented method with per-engine and per-topic counts. Preprint.",
    "evidence_type": "paper",
    "publisher": "Mowafak Allaham and Nicholas Diakopoulos (Northwestern University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Six commercial chatbots exceeded 90% multiple-choice accuracy on 2,100 same-day BBC news questions but fell 11-17% under free response; retrieval failures caused over 70% of errors.",
    "description": "Stanford researchers evaluated Gemini 3 Flash/Pro, Grok 4, Claude 4.5 Sonnet, GPT-5 and GPT-4o mini as news intermediaries across six BBC regional services (US & Canada, Arabic, Afrique, Hindi, Russian, Turkish) over February 9-22, 2026. Hindi queries scored lowest (79% vs 89-91%), models showed an Anglophone retrieval bias (citing English Wikipedia more than any Hindi outlet for Hindi queries), and the most vulnerable model accepted fabricated premises 64% of the time.",
    "date": "2026-05-21",
    "application_area": "Citations & Source Selection",
    "source": "arXiv: Evaluating Commercial AI Chatbots as News Intermediaries",
    "source_link": "https://arxiv.org/abs/2605.22785",
    "credibility": "High",
    "credibility_rationale": "Authored by Stanford researchers including Dan Jurafsky, James Zou, Daniel E. Ho and Mirac Suzgun, with a clearly documented method (2,100 same-day BBC questions across six regional services, six commercial chatbots) and covered by Stanford HAI and SSRC MediaWell. Caveat: arXiv preprint, not yet peer-reviewed.",
    "evidence_type": "paper",
    "publisher": "arXiv (Stanford University researchers: Suzgun, Shen, Bianchi, Spangher, Icard, Ho, Jurafsky, Zou)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Earned media accounts for 84% of AI citations and journalism for 27%, paid/advertorial content only 0.3%; ChatGPT cites sources in 96% of responses, Gemini 82%, Claude 55%.",
    "description": "Muck Rack's May 2026 'What Is AI Reading?' analyzed 25M+ links from ChatGPT, Claude and Gemini across 17 industries; across three editions earned media ranged 82-89%. ChatGPT averages 5 citations per response, Gemini 8, Claude 13. Muck Rack's 'earned media' bucket includes Wikipedia, academic, government and UGC sources, not only journalism.",
    "date": "2026-05-07",
    "application_area": "Citations & Source Selection",
    "source": "Muck Rack: What Is AI Reading? (May 2026) - Earned media still drives 84% of AI citations",
    "source_link": "https://muckrack.com/blog/what-is-ai-reading-may-2026",
    "credibility": "High",
    "credibility_rationale": "Major PR-software/data vendor with a large tracked series covered by PR Daily; PR Daily's later critique disputes the 'earned media' label but not the numbers. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "Muck Rack",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Wikipedia (13.15%) and Reddit (11.97%) make up over 25% of U.S. ChatGPT citations, while WSJ, NYT, Bloomberg and FT are absent from the top 20 (Forbes #18 at 1.38%).",
    "description": "5W Research's Citation Source Audit Q1 2026 synthesizes eleven published datasets; the headline leaderboard uses Similarweb's analysis of ~600,000 U.S. citation events across ChatGPT and Google AI Mode (Jan-Feb 2026; Similarweb report dated April 14, 2026, showing Reuters #7 at 2.27%). 5W ran no primary research.",
    "date": "2026-05",
    "application_area": "Citations & Source Selection",
    "source": "5W Research: Citation Source Audit Q1 2026",
    "source_link": "https://www.5wpr.com/research/citation-source-audit-q1-2026/",
    "credibility": "Medium",
    "credibility_rationale": "Agency secondary synthesis without named authors; the leaderboard figures were verified by the qualifier against Similarweb's primary report, but that primary page could not be re-located during leader review.",
    "evidence_type": "analysis",
    "publisher": "5W Public Relations (5W Research), reporting Similarweb data",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Average links per Google AI Overview rose from 6.82 (Nov 2024) to 15.22 (Feb 2026), while AIO presence reached 59.73% of tracked queries, up from ~28% in May 2025.",
    "description": "SE Ranking's continuously updated AI Overviews tracking page (last updated April 28, 2026) reports ten-word queries trigger AIOs over five times more often than single-word searches (69.21% vs 12.78%), informational queries account for over 96% of AIOs and transactional only 1.2%, YouTube took 10.74% and Reddit 4.01% of AIO citations in February 2026, and the Finance AIO appearance rate rose from 11% (May 2025) to 78% (February 2026).",
    "date": "2026-04-28",
    "application_area": "Citations & Source Selection",
    "source": "SE Ranking: Google's AI Overviews: Updates and changes from SGE to now",
    "source_link": "https://seranking.com/blog/ai-overviews/",
    "credibility": "Medium",
    "credibility_rationale": "SE Ranking is an established SEO data vendor with an ongoing AIO research program cited in third-party roundups, and all figures were verified on the page. However this continuously updated overview article does not state the keyword sample size or full methodology for its monthly tracking numbers, and its AIO presence figures diverge substantially from Semrush's 10M-keyword tracking, so Medium rather than High.",
    "evidence_type": "study",
    "publisher": "SE Ranking (Yulia Deda; reviewed by Svitlana Tomko, SEO Research Analyst)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Fandom.com is Google AI Mode's most-cited domain (7.16%), ahead of Wikipedia 5.21%, YouTube 4.91% and Reddit 4.19%, across ~600,000 US citation events (Jan-Feb 2026).",
    "description": "Similarweb analyzed nearly 600,000 citation events (an AI engine linking to a specific URL in a response) across ChatGPT web-browsing mode and Google AI Mode in the US during January-February 2026. AI Mode also cited google.com 2.85%, facebook.com 2.44%, amazon.com 1.84%, nih.gov 1.49%, github.com 1.48% and apple.com 1.37%; in ChatGPT, openai.com (6.21%), walmart.com (2.90%), youtube.com (2.67%), linkedin.com (2.42%) and reuters.com (2.27%) followed Wikipedia and Reddit.",
    "date": "2026-04-14",
    "application_area": "Citations & Source Selection",
    "source": "Similarweb AI Search Blog: Most Cited Domains in LLMs (ChatGPT and Google AI Mode)",
    "source_link": "https://aisearch.similarweb.com/blog/most-cited-domains-llms/",
    "credibility": "High",
    "credibility_rationale": "Similarweb is a large, publicly listed digital-intelligence data vendor. The post documents its method: nearly 600,000 citation events across ChatGPT web-browsing mode and Google AI Mode, US, January-February 2026, with a defined unit of analysis and raw counts published alongside percentages.",
    "evidence_type": "analysis",
    "publisher": "Similarweb",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Over 11,000 queries, SearchGPT's top-100 cited domains overlapped 24-25% with organic Google vs 68% for AI Overviews; Wikipedia was cited for 49% of SearchGPT queries.",
    "description": "'Answer Bubbles' (Huang, Goyal, Saha, Chandrasekharan; UIUC; EMNLP 2026) compared Vanilla GPT, SearchGPT, Google AI Overviews, Perplexity Search with Grok and organic Google on 11,000 real queries. Wikipedia was the most cited domain everywhere (organic Google 81% of queries, SearchGPT 49%, AIO 28%) and was further over-represented in generated summaries (+2.6 pp SearchGPT, +5.4 pp AIO); SearchGPT drew 0.1% of citations from social platforms versus 8.5% for AIO and 13.4% for organic.",
    "date": "2026-03-17",
    "application_area": "Citations & Source Selection",
    "source": "arXiv / EMNLP 2026: Answer Bubbles: Information Exposure in AI-Mediated Search (arXiv 2603.16138)",
    "source_link": "https://arxiv.org/html/2603.16138v2",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (arXiv comments state EMNLP 2026). Authors are at UIUC's Siebel School of Computing and Data Science; Chandrasekharan and Saha are established computational social science researchers. Large sample (11,000 real queries across five systems) with documented methodology for source diversity, summary linguistics and source-summary fidelity.",
    "evidence_type": "paper",
    "publisher": "University of Illinois Urbana-Champaign (Siebel School of Computing and Data Science) via arXiv; accepted to EMNLP 2026",
    "publisher_type": "research-institution"
  },
  {
    "finding": "ChatGPT left 85% of the 548,534 pages it retrieved uncited; pages ranking #1 in Google were cited 43.2% of the time, 3.5x more than pages outside Google's top 20.",
    "description": "AirOps analyzed 15,000 queries (43,233 incl. fan-outs), 548,534 retrieved pages and 82,108 ChatGPT citations. 55.8% of cited pages ranked in Google's top 20 for some query; 32.9% appeared only for a fan-out query; ~74% of citations went to sites with DA under 80.",
    "date": "2026-03-12",
    "application_area": "Citations & Source Selection",
    "source": "AirOps: The Influence of Retrieval, Fan-out, and Google SERPs on ChatGPT Citations",
    "source_link": "https://www.airops.com/report/influence-of-retrieval-fanout-and-google-serps-in-chatgpt",
    "credibility": "Medium",
    "credibility_rationale": "AI content/GEO vendor, but large-sample and methodology-documented; no independent corroboration verified.",
    "evidence_type": "study",
    "publisher": "AirOps (Oshen Davidson)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Only 37.9% of URLs cited in Google AI Overviews rank in the organic top 10, down from 76.1% in July 2025; 31.2% rank 11-100 and 31.0% rank beyond the top 100.",
    "description": "Ahrefs analyzed 863K keyword SERPs and 4M AI Overview URLs (double its July 2025 study). Organic blue links only: 37.1% top 10, 26.2% positions 11-100, 36.7% beyond 100; YouTube was 18.2% of non-ranking citations. Ahrefs attributes the shift to query fan-out.",
    "date": "2026-03-02",
    "application_area": "Citations & Source Selection",
    "source": "Ahrefs: Update: 38% of AI Overview Citations Pull From The Top 10",
    "source_link": "https://ahrefs.com/blog/ai-overview-citations-top-10/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor; large documented sample (863K SERPs, 4M URLs) updating a documented prior study; covered by Search Engine Journal and DesignRush. Author Louise Linehan. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "88% of Google AI Mode citations are not in the organic top 10 for the exact-match query, and 96% of AI Mode responses include at least one citation.",
    "description": "Moz studied nearly 40,000 queries (US and UK, desktop and mobile, via STAT), extracting traditional SERPs to measure overlap with AI Mode citations. Most responses pulled from 10+ unique URLs; YouTube was the second most cited external source; the top 4 external sources took only 10% of citations. Moz's LinkedIn summary (linkedin.com/posts/moz_our-study-of-40000-queries-found-that-88-activity-7432761636597362689-1XAv) confirms the figures.",
    "date": "2026-02-18",
    "application_area": "Citations & Source Selection",
    "source": "Moz: Only 12% of AI Mode Citations Match URLs in the Organic SERP",
    "source_link": "https://moz.com/blog/ai-mode-citations",
    "credibility": "High",
    "credibility_rationale": "Long-established SEO data vendor; full report (Chima Mmeje, Feb 18, 2026) documents the ~40,000-query STAT method and links the dataset; cited by The Next Web and multiple SEO sites. Spot-check via Moz's LinkedIn post confirmed 88%, 96% and 40,000 queries (moz.com blocked for the fetch tool).",
    "evidence_type": "study",
    "publisher": "Moz",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Across 55,936 queries and 1.4M citation links, LLM search engines cite more diverse domains than Google/Bing (37% unique to LLM engines) but are no better on credibility or neutrality.",
    "description": "Zhang et al. (HKUST Guangzhou / Rutgers) compared six LLM-based engines with Google and Bing on 55,936 queries covering 124,287 domains and 1,418,733 citation links. 37% of cited domains were unique to LLM engines; no advantage on credibility, political neutrality or safety. Preprint.",
    "date": "2025-12-10",
    "application_area": "Citations & Source Selection",
    "source": "arXiv (HKUST Guangzhou / Rutgers): Source Coverage and Citation Bias in LLM-based vs. Traditional Search Engines",
    "source_link": "https://arxiv.org/abs/2512.09483",
    "credibility": "High",
    "credibility_rationale": "University-authored large-scale measurement (Gareth Tyson, Kiran Garimella are established web-measurement researchers) with described method. Preprint, not yet peer-reviewed.",
    "evidence_type": "paper",
    "publisher": "Peixian Zhang, Qiming Ye, Zifan Peng, Gareth Tyson (HKUST Guangzhou); Kiran Garimella (Rutgers University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "LLMs used as citation selectors cite left-leaning outlets at notably higher rates than BM25 or dense retrievers, driven by outlet-name recognition rather than article content.",
    "description": "Dai, Cao, Wang, Pang, Xu, Ng and Chua (EMNLP 2025) built AllSides-2024 (real news articles from January-December 2024 labeled left/right) and compared LLM citation choices to BM25 and dense retrievers. Controlled experiments showed the bias arises from a preference for outlets identified as left-leaning rather than left-oriented content: LLMs almost perfectly recognize an outlet's orientation from its name but struggle to infer bias from content alone, removing the source name significantly reduced the Citation Preference Index and swapping names reversed the bias, so publisher identity carries citation weight independent of the text.",
    "date": "2025-11",
    "application_area": "Citations & Source Selection",
    "source": "ACL Anthology (EMNLP 2025): Media Source Matters More Than Content: Unveiling Political Bias in LLM-Generated Citations",
    "source_link": "https://aclanthology.org/2025.emnlp-main.872/",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed main-conference paper at EMNLP 2025 (ACL Anthology, pages 17256-17276) by established IR/NLP researchers from Renmin University of China, Chinese Academy of Sciences and National University of Singapore, with a newly constructed dataset, a Citation Preference Index metric, retriever baselines and controlled name-removal and name-swap ablations.",
    "evidence_type": "paper",
    "publisher": "Association for Computational Linguistics, Proceedings of EMNLP 2025 (Dai, Cao, Wang, Pang, Xu, Ng, Chua)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Across 2,709 AI assistant news responses in 18 countries, 45% had a significant issue and 31% had sourcing problems; Gemini had sourcing issues in 72% of responses.",
    "description": "Coordinated by the EBU and led by the BBC, journalists at 22 public service media organisations across 18 countries and 14 languages rated free-tier ChatGPT, Copilot, Gemini and Perplexity responses to 30 core news questions (generated late May to early June 2025) on accuracy, sourcing, opinion vs fact, editorialisation and context. 20% of responses had major accuracy issues; other assistants had significant sourcing issues in under 25% of responses, and errors were systemic across all languages and assistants.",
    "date": "2025-10-21",
    "application_area": "Citations & Source Selection",
    "source": "EBU / BBC: News Integrity in AI Assistants - An international PSM study",
    "source_link": "https://www.ebu.ch/files/live/sites/ebu/files/Publications/MIS/open/EBU-MIS-BBC_News_Integrity_in_AI_Assistants_Report_2025.pdf",
    "credibility": "High",
    "credibility_rationale": "Large multi-organisation primary study coordinated by the EBU and led by the BBC: 22 public service media organisations across 18 countries and 14 languages, 2,709 responses evaluated by journalists, with detailed methodology, per-assistant sample sizes (Copilot n=675, ChatGPT n=678, Perplexity n=681, Gemini n=675) and data tables. Widely covered by Reuters, Forbes, Al Jazeera and The Register.",
    "evidence_type": "study",
    "publisher": "European Broadcasting Union (EBU) with BBC",
    "publisher_type": "research-institution"
  },
  {
    "finding": "53% of domains cited by Google AI Overviews fall outside the top-10 organic results and 27% outside the top 100; AIO citations overlapped only 18% across two months vs 45% for organic.",
    "description": "Kirsten et al. (MPI-SWS / Ruhr University Bochum, Findings of ACL 2026) compared Google organic with five generative systems on 4,706 queries across six datasets. Perplexity Sonar cited the least popular domains (median Tranco 5,647 vs 2,352); 9-27% of binary questions flipped answer polarity within five minutes across engines.",
    "date": "2025-10-13",
    "application_area": "Citations & Source Selection",
    "source": "arXiv / Findings of ACL 2026 (Max Planck Institute for Software Systems, Ruhr University Bochum): Characterizing Web Search in The Age of Generative AI",
    "source_link": "https://arxiv.org/html/2510.11560v2",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (Findings of ACL 2026); MPI-SWS (Krishna Gummadi) and RUB authors; documented method; figures verified in the full text.",
    "evidence_type": "paper",
    "publisher": "Elisabeth Kirsten, Jost Grosse Perdekamp, Qinyuan Wu, Mihir Upadhyay, Krishna P. Gummadi, Muhammad Bilal Zafar (MPI-SWS / Ruhr University Bochum)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Across 6.8M AI citations, 86% came from brand-manageable sources (websites 44%, listings 42%); Gemini drew 52.15% from brand websites, OpenAI 48.73% from third-party listings.",
    "description": "Yext Scout analyzed 6.8 million citations from about 1.6 million responses across Gemini, OpenAI and Perplexity between July 1 and August 31, 2025, spanning 20,820 domains and 200,000+ locations in retail, financial services, healthcare and food service using branded/unbranded x objective/subjective query quadrants. Reviews and social were 8% of citations, news/forums/other about 6% and forums just over 2%; Perplexity diversified (364K+ MapQuest citations), financial services (48.16%) and retail (47.62%) drew most from brand websites, healthcare 52.61% from listings, and food service 41.63% from listings and 13.28% from reviews/social.",
    "date": "2025-10-09",
    "application_area": "Citations & Source Selection",
    "source": "Yext Research: AI Citations, User Locations, & Query Context",
    "source_link": "https://www.yext.com/research/article/ai-citations-user-locations-query-context",
    "credibility": "High",
    "credibility_rationale": "Yext is a publicly traded digital presence platform; the study covers 6.8 million citations from 1.6 million responses across three engines, 20,820 domains and 200,000+ locations with a documented query framework and named authors, and its press release was carried by Business Wire, Morningstar, Nasdaq and Yahoo Finance. Caveat: Yext sells listings management, so the emphasis on listings aligns with its commercial interest.",
    "evidence_type": "study",
    "publisher": "Yext",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Official party websites made up only ~25-45% of citations in US political AI answers vs 60-65% in Japan, and low-barrier sources were often cited yet poorly reflected in answers.",
    "description": "Mochizuki, Komatsu, Noguchi and Ataka propose evaluation criteria based on citation publisher attributes to estimate the content-injection barrier, i.e. how easily an attacker could poison an answer via web content. Comparing US and Japanese political queries, the lower share of primary-source citations in the US indicates higher poisoning risk, and sources with low content-injection barriers are frequently cited yet poorly reflected in answer content.",
    "date": "2025-10-08",
    "application_area": "Citations & Source Selection",
    "source": "arXiv: Exposing Citation Vulnerabilities in Generative Engines (2510.06823)",
    "source_link": "https://arxiv.org/abs/2510.06823",
    "credibility": "Medium",
    "credibility_rationale": "Academic arXiv preprint (cs.CR) by a four-author Japanese research team with stated numbers verified on the abstract page. However the paper is listed as under review with no peer-reviewed acceptance, author affiliations are not shown, and the experiment is scoped to political queries in two countries with a modest citation footprint, so Medium rather than High.",
    "evidence_type": "paper",
    "publisher": "arXiv (Riku Mochizuki, Shusuke Komatsu, Souta Noguchi, Kazuto Ataka)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Overlap between Google AI Overview citations and organic rankings rose from 32.3% to 54.5% over 16 months, reaching 68-75% in YMYL sectors but only 22.9% in E-commerce.",
    "description": "BrightEdge Generative Parser tracked citation/rank overlap across nine industries, May 2024-Sept 2025. Healthcare 75.3%, Education 72.6%, B2B Tech 71.0%, Insurance 68.6%; Finance 32.2%, Travel 23.6%, E-commerce 22.9%, Restaurants 19.2%. Measures any-rank overlap (only 16.7% from the top 10), so it is not directly comparable with Ahrefs' 2026 top-10 figure.",
    "date": "2025-09-18",
    "application_area": "Citations & Source Selection",
    "source": "BrightEdge: AI Overview Citations Now 54% from Organic Rankings",
    "source_link": "https://www.brightedge.com/resources/weekly-ai-search-insights/rank-overlap-after-16-months-of-aio",
    "credibility": "High",
    "credibility_rationale": "Large enterprise SEO data vendor; tool, industries and 16-month window documented; covered by Search Engine Journal and reconciled against other studies by third parties. Exact query counts not stated.",
    "evidence_type": "study",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Overlap between Google organic results and AI search citations is low: 15-33% at the top 5 and 20% to 50%+ at the top 10, depending on product category.",
    "description": "University of Toronto researchers ran 1,000 consumer ranking prompts (10 categories x 100), five target languages and seven paraphrase templates. Electric Cars showed the highest overlap (33% top 5, 50%+ top 10); Smartphones and Laptops 15% and 32% at top 5. Large engine differences in domain diversity, freshness, cross-language consistency and phrasing sensitivity.",
    "date": "2025-09-10",
    "application_area": "Citations & Source Selection",
    "source": "arXiv (University of Toronto): Generative Engine Optimization: How to Dominate AI Search",
    "source_link": "https://arxiv.org/abs/2509.08919",
    "credibility": "High",
    "credibility_rationale": "Four University of Toronto researchers including Prof. Nick Koudas; documented methodology, indexed by dblp and Semantic Scholar; preprint, not yet peer-reviewed.",
    "evidence_type": "paper",
    "publisher": "Mahe Chen, Xiaoxuan Wang, Kaiwen Chen, Nick Koudas (University of Toronto)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "The 10 leading AI chatbots repeated false news claims 35% of the time in August 2025, nearly double the 18% of August 2024, as their non-response rate fell from 31% to 0%.",
    "description": "NewsGuard's one-year AI False Claims Monitor progress report, the first to name individual models (ChatGPT-5, You.com, Grok, Pi, le Chat, Copilot, Meta AI, Claude, Gemini, Perplexity), found that as chatbots gained real-time web search and stopped declining to answer, they increasingly treated low-quality sites and AI content farms as credible sources.",
    "date": "2025-09-04",
    "application_area": "Citations & Source Selection",
    "source": "NewsGuard: One-Year AI Audit Progress Report Finds that AI Models Spread Falsehoods in the News 35% of the Time",
    "source_link": "https://www.newsguardtech.com/press/newsguard-one-year-ai-audit-progress-report-finds-that-ai-models-spread-falsehoods-in-the-news-35-of-the-time/",
    "credibility": "Medium",
    "credibility_rationale": "NewsGuard is an established misinformation-ratings company whose monthly AI False Claims Monitor is regularly cited by Reuters, Axios, Forbes and academic work, and it publishes a methodology page. However this is a press release from a commercial vendor, the sample is small (10 chatbots x 10 false claims per month x three prompt styles) and ratings involve subjective analyst judgement, so it does not reach the bar for High.",
    "evidence_type": "study",
    "publisher": "NewsGuard",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "ChatGPT's cited URLs overlap with Google's top 10 only 10.00% of the time (domains 31.80%), versus 65.07% URL and 80.58% domain overlap for Perplexity.",
    "description": "Ahrefs tested about 3,311 short-tail keywords across ChatGPT, Perplexity and Google's top 100. ChatGPT cites domains roughly 3x more often than the specific ranking page, its fan-out query overlap with Google's top 10 was lowest at 6.82% (long-tail 7.05%), and Ahrefs concludes ChatGPT applies additional processing after retrieving Google results, making a Google ranking a much weaker predictor of a ChatGPT citation than of a Perplexity citation.",
    "date": "2025-09-03",
    "application_area": "Citations & Source Selection",
    "source": "Ahrefs: ChatGPT May Scrape Google, but the Results Don't Match",
    "source_link": "https://ahrefs.com/blog/chatgpt-google-citations",
    "credibility": "High",
    "credibility_rationale": "First-party Ahrefs study (author Louise Linehan, reviewed by Ryan Law) with a stated sample of roughly 3,311 short-tail keywords plus long-tail and fan-out query sets, comparing ChatGPT and Perplexity citations against Google's top 10 and top 100 with exact overlap percentages and the comparison method described.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Only 12% of URLs cited by AI assistants rank in Google's top 10 for the original prompt: Perplexity 28.6%, Gemini 8.6%, Copilot 8.2%, ChatGPT 8% (15,000 queries).",
    "description": "Ahrefs ran 15,000 long-tail queries through Google, Bing, ChatGPT, Gemini, Copilot and Perplexity and classified cited URLs by rank. ChatGPT in-text citations matched Google's top 10 8% of the time and reference citations 6.1%; against Bing's top 10 the average was 10% (Copilot 16.6%, Perplexity 14%), about 80% of citations did not rank anywhere in Google for the original query, versus 76% top-10 alignment for Google's own AI Overviews in Ahrefs' earlier study.",
    "date": "2025-08-11",
    "application_area": "Citations & Source Selection",
    "source": "Ahrefs: Only 12% of AI Cited URLs Rank in Google's Top 10 for the Original Prompt",
    "source_link": "https://ahrefs.com/blog/ai-search-overlap/",
    "credibility": "High",
    "credibility_rationale": "Ahrefs is a major SEO data vendor; the study by data scientist Xibeijia Guan uses 15,000 long-tail queries from Ahrefs Brand Radar with a documented methodology (queries run in Google and Bing, then asked of four assistants; cited URLs classified by top-10/top-100 rank) and averages shown with their arithmetic on the page.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "The top 20 news outlets account for 67.3% of OpenAI models' news citations (Gini 0.83) vs 31.9% for Google and 28.5% for Perplexity; Reuters alone is 22.8% of OpenAI's.",
    "description": "Kai-Cheng Yang (Northeastern) analyzed 24,069 conversations and over 366,000 citations (32,865 news citations from 1,795 news domains, 9% of the total) from 12 AI search models across OpenAI, Perplexity and Google. OpenAI models leaned on Reuters (22.8%), AP News (12.2%), Financial Times (7.0%) and Axios (6.7%); Google on India Times (4.3%), Forbes (2.5%) and Al Jazeera (2.5%); Perplexity on BBC (3.2%) and Yahoo (2.7%). Center-rated sources dominated (OpenAI 85.2%, Google 75.2%, Perplexity 72.8%), high-quality sources made up 89.7-96.2% of news citations and right-leaning outlets 0.3-1.2%.",
    "date": "2025-07-07",
    "application_area": "Citations & Source Selection",
    "source": "arXiv: News Source Citing Patterns in AI Search Systems (arXiv 2507.05301)",
    "source_link": "https://arxiv.org/html/2507.05301v1",
    "credibility": "High",
    "credibility_rationale": "Single-author university preprint, but Kai-Cheng Yang (Northeastern University) is a well-cited computational social scientist, and the sample is large and documented: 24,069 conversations, over 366,000 citations, 32,865 news citations across 12 models, using published third-party outlet ratings for leaning and quality. Not peer reviewed on the fetched version.",
    "evidence_type": "paper",
    "publisher": "Northeastern University (Kai-Cheng Yang) via arXiv",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Wikipedia is the most-cited domain in ChatGPT (16.3%), Perplexity (12.5%) and Google AI Overviews (8.4%); ChatGPT leans on news (Reuters 4%), Perplexity on YouTube (16.1%).",
    "description": "Ahrefs (Patrick Stox) used Brand Radar data for June 2025 (~76.7M AIOs, 957K ChatGPT prompts, 953.5K Perplexity prompts). AIOs favored UGC (YouTube 9.5%, Reddit 7.4%, Quora), ChatGPT favored news publishers, Perplexity favored video.",
    "date": "2025-06-11",
    "application_area": "Citations & Source Selection",
    "source": "Ahrefs: Top 10 Most Cited Domains by AI Assistants",
    "source_link": "https://ahrefs.com/blog/top-10-most-cited-domains-ai-assistants",
    "credibility": "High",
    "credibility_rationale": "Major data vendor; large Brand Radar sample stated; authored by a widely cited SEO researcher and reviewed by Ryan Law; figures confirmed on the page.",
    "evidence_type": "study",
    "publisher": "Ahrefs (Patrick Stox; reviewed by Ryan Law)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Across 680M citations, Wikipedia is ChatGPT's top source (7.8% of citations, 47.9% of its top-10 share) while Reddit leads Perplexity (6.6%, 46.7% of top-10) and Google AIO (2.2%).",
    "description": "Profound analyzed 680 million citations across ChatGPT, Google AI Overviews and Perplexity, Aug 2024-Jun 2025. ChatGPT leans on Wikipedia, AIO on Reddit/YouTube/Quora/LinkedIn with the most balanced distribution, Perplexity concentrates on Reddit; .com domains were 80.41% of ChatGPT citations.",
    "date": "2025-06-05",
    "application_area": "Citations & Source Selection",
    "source": "Profound: AI Platform Citation Patterns: How ChatGPT, Google AI Overviews, and Perplexity Source Information",
    "source_link": "https://www.tryprofound.com/blog/ai-platform-citation-patterns",
    "credibility": "High",
    "credibility_rationale": "Major AI-visibility data vendor; very large sample with platform breakdowns; cited by Entrepreneur, Tinuiti, PR Newswire and 5WPR. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Profound",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Users prefer search-augmented LLM answers with more citations (beta = 0.209) even when citations do not support the claims; citing Wikipedia is negatively associated (-0.071).",
    "description": "Search Arena (UC Berkeley/LMArena, ICLR 2026): 24,000+ paired interactions and ~12,000 preference votes across 13 search-augmented LLMs. Supporting (0.285) and irrelevant (0.273) citations both correlated positively with preference; tech/code platforms (0.073), community (0.061) and social (0.057) positive, Wikipedia negative (-0.071).",
    "date": "2025-06-05",
    "application_area": "Citations & Source Selection",
    "source": "arXiv / ICLR 2026 (UC Berkeley, LMArena): Search Arena: Analyzing Search-Augmented LLMs",
    "source_link": "https://arxiv.org/abs/2506.05334",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (ICLR 2026 poster); UC Berkeley and LMArena authors; public code and dataset; large sample with documented regression; coefficients verified in the full text.",
    "evidence_type": "paper",
    "publisher": "Mihran Miroyan et al. (UC Berkeley / LMArena)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "75.7% of sources cited in Google AI Overviews rank for 1,000+ keywords, and 47% of queries cite identical domains across all five US locations tested.",
    "description": "SE Ranking analyzed 100,013 keywords across 20 niches in five US cities (data collected April 15, 2025). ~28% of queries triggered an AIO with 13.34 sources on average; 47.05% cited identical domains across locations; 43.42% of AIOs linked back to Google's own results pages.",
    "date": "2025-05-09",
    "application_area": "Citations & Source Selection",
    "source": "SE Ranking: AI Overviews Research: A State-by-State US Comparison",
    "source_link": "https://seranking.com/blog/ai-overviews-us-states-comparison-research/",
    "credibility": "High",
    "credibility_rationale": "Established SEO data vendor; large documented sample (100,013 keywords, 20 niches, 5 locations, dated capture); covered by Search Engine Journal and cited in stat roundups.",
    "evidence_type": "study",
    "publisher": "SE Ranking",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Eight AI search engines answered over 60% of 1,600 news-citation queries incorrectly; Perplexity was best (37% wrong) and Grok 3 worst (94% wrong).",
    "description": "Tow Center researchers gave ChatGPT Search, Perplexity, Perplexity Pro, DeepSeek Search, Copilot, Grok-2, Grok-3 and Gemini 200 verbatim excerpts from 20 publishers and asked for headline, publisher, date and URL. Paid tiers (Perplexity Pro, Grok 3) erred more often than free versions because they answered confidently instead of declining; ChatGPT got 134 of 200 wrong but hedged in only 15 responses.",
    "date": "2025-03-06",
    "application_area": "Citations & Source Selection",
    "source": "Columbia Journalism Review / Tow Center for Digital Journalism: AI Search Has a Citation Problem",
    "source_link": "https://www.cjr.org/tow_center/we-compared-eight-ai-search-engines-theyre-all-bad-at-citing-news.php",
    "credibility": "High",
    "credibility_rationale": "Primary research from the Tow Center for Digital Journalism at Columbia University (Jazwinska and Chandrasekar) with a documented method: 20 publishers x 10 articles x 8 chatbots = 1,600 queries, per-tool results reported. Widely cited by major outlets and by the BBC/EBU report.",
    "evidence_type": "study",
    "publisher": "Columbia Journalism Review / Tow Center for Digital Journalism (Columbia University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Grok 3 returned fabricated or broken URLs in 154 of 200 responses, and over half of the citations from both Gemini and Grok 3 led to error pages.",
    "description": "In the same 1,600-query Tow Center test of eight generative search engines, chatbots frequently generated non-existent or dead links rather than admitting they could not find the article. They also often cited syndicated copies on sites like Yahoo or AOL instead of the original publisher, even when the publisher had a licensing deal with the AI company.",
    "date": "2025-03-06",
    "application_area": "Citations & Source Selection",
    "source": "Columbia Journalism Review / Tow Center for Digital Journalism: AI Search Has a Citation Problem",
    "source_link": "https://www.cjr.org/tow_center/we-compared-eight-ai-search-engines-theyre-all-bad-at-citing-news.php",
    "credibility": "High",
    "credibility_rationale": "Primary research from the Tow Center for Digital Journalism at Columbia University (Jazwinska and Chandrasekar) with a documented method: 20 publishers x 10 articles x 8 chatbots = 1,600 queries, per-tool results reported. Widely cited by major outlets and by the BBC/EBU report.",
    "evidence_type": "study",
    "publisher": "Columbia Journalism Review / Tow Center for Digital Journalism (Columbia University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Roughly 70% of pages cited in Google AI Overviews change within 2-3 months; AIO citation volatility (0.68-0.73) far exceeds organic ranking volatility (0.49-0.55).",
    "description": "Authoritas tracked 11,203 categorized US desktop keywords captured on August 23, 2024, October 17, 2024 and January 17, 2025; AI Overviews appeared for 18.8% (2,104). AIO summary text was comparatively stable (volatility 0.18 at 8 weeks, 0.21 at 13 weeks) even as cited pages churned, and AIO citation changes were almost uncorrelated with SERP layout changes (0.04), so citation tracking needs frequent re-sampling.",
    "date": "2025-02-19",
    "application_area": "Citations & Source Selection",
    "source": "Authoritas: SERP Organic and AI Overview Volatility Research",
    "source_link": "https://www.authoritas.com/blog/serp-organic-and-ai-overview-volatility-research",
    "credibility": "Medium",
    "credibility_rationale": "SEO platform vendor study by CEO Laurence O'Toole with a stated sample (11,203 US desktop keywords, three captures) and defined volatility metrics; Authoritas has a track record of AIO research cited by Search Engine Land. Not High because the sample is moderate, the volatility scoring method is proprietary, and it is single-vendor, non-peer-reviewed work.",
    "evidence_type": "study",
    "publisher": "Authoritas",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "51% of AI assistant answers to 100 BBC news questions had significant issues and 91% had some issue; 19% of answers citing BBC content introduced factual errors.",
    "description": "BBC journalists reviewed responses from ChatGPT, Copilot, Gemini and Perplexity (generated 5-6 December 2024 with BBC's crawler blocks temporarily lifted) against seven criteria including accuracy, attribution and representation of BBC content. 13% of BBC quotes were altered or absent from the cited article; Gemini had the highest rate of responses with significant issues (over 60%) and 46% of Gemini responses were flagged for significant accuracy issues.",
    "date": "2025-02",
    "application_area": "Citations & Source Selection",
    "source": "BBC: Representation of BBC News content in AI Assistants",
    "source_link": "https://www.bbc.co.uk/aboutthebbc/documents/bbc-research-into-ai-assistants.pdf",
    "credibility": "High",
    "credibility_rationale": "Primary research report published by the BBC (24-page PDF, February 2025) with a full methodology appendix: 100 news questions, responses collected 5-6 December 2024 with robots.txt blocks lifted, seven rating criteria, BBC journalists as expert reviewers, product versions listed. Caveat: the BBC is the subject of the content tested and has an institutional interest.",
    "evidence_type": "study",
    "publisher": "BBC",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "Prompted to use BBC sources, Perplexity cited BBC in 100% of responses, ChatGPT and Copilot in 70% and Gemini in 53%; 26% of Gemini answers gave no sources at all.",
    "description": "In the BBC's 100-question study of ChatGPT, Copilot, Gemini and Perplexity, assistants often chose old articles or live pages as sources, producing outdated claims such as that Rishi Sunak and Nicola Sturgeon were still in office. Over 45% of Gemini responses had significant sourcing errors, 7% of ChatGPT responses gave no sources, and significant issues with representation of BBC content were found in 34% of Gemini, 27% of Copilot, 17% of Perplexity and 15% of ChatGPT responses.",
    "date": "2025-02",
    "application_area": "Citations & Source Selection",
    "source": "BBC: Representation of BBC News content in AI Assistants",
    "source_link": "https://www.bbc.co.uk/aboutthebbc/documents/bbc-research-into-ai-assistants.pdf",
    "credibility": "High",
    "credibility_rationale": "Primary research report published by the BBC (24-page PDF, February 2025) with a full methodology appendix: 100 news questions, responses collected 5-6 December 2024 with robots.txt blocks lifted, seven rating criteria, BBC journalists as expert reviewers, product versions listed. Caveat: the BBC is the subject of the content tested and has an institutional interest.",
    "evidence_type": "study",
    "publisher": "BBC",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "ChatGPT Search gave partially or entirely wrong attributions for 153 of 200 publisher quotes and admitted it could not answer only 7 times.",
    "description": "Tow Center tested ChatGPT Search with 200 quotes from 20 publishers spanning OpenAI licensing partners, litigants and unaffiliated outlets. Even licensed partners with all OpenAI crawlers enabled (New York Post, The Atlantic) were frequently miscited; when blocked from the New York Times, ChatGPT attributed a Times quote to a site that had plagiarized the article, and repeated identical queries produced inconsistent citations.",
    "date": "2024-11-27",
    "application_area": "Citations & Source Selection",
    "source": "Columbia Journalism Review / Tow Center for Digital Journalism: How ChatGPT Search (Mis)represents Publisher Content",
    "source_link": "https://www.cjr.org/tow_center/how-chatgpt-misrepresents-publisher-content.php",
    "credibility": "High",
    "credibility_rationale": "Primary study by the Tow Center at Columbia University (Jazwinska and Chandrasekar) with a stated method: 200 quotes, 10 each from 20 publishers spanning licensing partners, litigants and unaffiliated outlets, with results and examples reported in detail. Widely cited by news and industry press.",
    "evidence_type": "study",
    "publisher": "Columbia Journalism Review / Tow Center for Digital Journalism (Columbia University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "A tree-of-attacks prompt injection on a product page lifted low-ranked products 54-96% of the maximum rank-score gain across five LLMs and ~3 positions on live Perplexity.",
    "description": "Pfrommer, Bai, Gautam and Sojoudi (UC Berkeley, EMNLP 2024) formalized conversational-search ranking as an adversarial problem on the RAGDOLL dataset (50 product categories, 8 real brands per category). Their Tree-of-Attacks-with-Pruning jailbreak raised the promoted product's mean ranking score by +3.38 on GPT-3.5 Turbo (57.53% of the gap to maximum), +5.00 on GPT-4 Turbo (82.94%), +6.02 on Llama 3 70B (95.74%), +4.13 on Mixtral 8x22 (76.23%) and +2.89 on Perplexity's Sonar Large Online (54.23%); attacks transferred to perplexity.ai on self-hosted sites closed more than half the gap to the top rank.",
    "date": "2024-06-05",
    "application_area": "Citations & Source Selection",
    "source": "Pfrommer et al. (UC Berkeley, EMNLP 2024): Ranking Manipulation for Conversational Search Engines",
    "source_link": "https://arxiv.org/html/2406.03589v1",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed EMNLP 2024 Main-track paper by UC Berkeley EECS researchers, with a released dataset (RAGDOLL, CC-BY-4.0), explicit experimental protocol, per-model results table and a transfer demonstration on a production system.",
    "evidence_type": "paper",
    "publisher": "arXiv / EMNLP 2024 (UC Berkeley EECS)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "75% of pages cited in Google AI Overviews ranked in the top 12 organic positions (median rank 4) across 120,778 US keywords, and 90% came from position 35 or higher.",
    "description": "Botify and DemandSphere analyzed 120,778 keywords across 22 websites (10 e-commerce retailers, 10 brands, two publishers) on US SERPs collected August 15 to September 1, 2024. AIOs appeared for 47% of keywords (59% of 85,638 informational vs 19% of 36,000 commercial), long-tail queries of 5+ words accounted for 73.6% of AIO triggers, and cosine similarity between cited pages and the AIO summary correlated with citation; the keyword set was curated to encourage AIO appearance, so 47% overstates typical incidence.",
    "date": "2024",
    "application_area": "Citations & Source Selection",
    "source": "Botify x DemandSphere: AI Overviews Study - Inside Google's New Search Reality",
    "source_link": "https://lp.botify.com/hubfs/White%20paper/2024/Botify%20x%20DemandSphere%20-%20AI%20Overviews%20Report.pdf",
    "credibility": "Medium",
    "credibility_rationale": "Joint white paper from two established enterprise SEO platforms with a large documented sample (120,778 US keywords, 57,263 AIO SERPs, 22 websites) and described methods, cited by Search Engine Land. Rated Medium because the keyword set was deliberately curated to encourage AIO appearance, pixel measurements were desktop-only, and CTR-impact claims rest on internal, undocumented Botify data.",
    "evidence_type": "study",
    "publisher": "Botify and DemandSphere",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "GEO tactics help lower-ranked pages most: the Cite Sources method lifted visibility of rank-5 source pages 115.1% while rank-1 pages lost 30.3%.",
    "description": "Same KDD 2024 GEO paper, Table 2: optimizing all sources simultaneously, Cite Sources gave +115.1% for rank-5 pages vs -30.3% for rank-1 (Quotation +99.7%, Statistics +97.9% for rank 5). Table 3 shows the best tactic varies by domain (e.g., Cite Sources for Facts/Law & Government, Quotation Addition for People & Society/History).",
    "date": "2023-11-16",
    "application_area": "Citations & Source Selection",
    "source": "arXiv / KDD 2024 (Princeton, IIT Delhi): GEO: Generative Engine Optimization (full text, v3)",
    "source_link": "https://arxiv.org/html/2311.09735v3",
    "credibility": "High",
    "credibility_rationale": "Same peer-reviewed KDD 2024 paper; Tables 2 and 3 confirmed in the full text; figures widely repeated by third parties.",
    "evidence_type": "paper",
    "publisher": "Pranjal Aggarwal et al. (Princeton University / IIT Delhi); ACM KDD 2024",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Only 51.5% of sentences from commercial generative search engines were fully supported by their citations, and only 74.5% of citations supported the attached claim.",
    "description": "Liu, Zhang and Liang (Stanford, Findings of EMNLP 2023) ran a human-evaluation audit of Bing Chat, NeevaAI, Perplexity.ai and YouChat, measuring citation recall (51.5%) and precision (74.5%); responses were fluent but often contained unsupported statements.",
    "date": "2023-04-19",
    "application_area": "Citations & Source Selection",
    "source": "arXiv / Findings of EMNLP 2023 (Stanford): Evaluating Verifiability in Generative Search Engines",
    "source_link": "https://arxiv.org/abs/2304.09848",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (Findings of EMNLP 2023) by Stanford researchers including Percy Liang, with documented human-evaluation method and public code.",
    "evidence_type": "paper",
    "publisher": "Nelson F. Liu, Tianyi Zhang, Percy Liang (Stanford University); Findings of EMNLP 2023",
    "publisher_type": "research-institution"
  },
  {
    "finding": "With identical specs, LLM recommenders picked the well-known brand 100% of the time, but a rating edge under +0.1 stars broke the monopoly and authority-style claims were worth +0.17 rating points.",
    "description": "Chu and Hou tested GPT-4o-mini, Claude Sonnet and Gemini 3 Flash on skincare recommendation with controlled catalogs (robustness checks on search goods). Established brands were recommended 100% of the time under identical specifications (Incumbent Advantage Index 10.0); unverified authority-style marketing claims achieved a 'Bias Surplus Value' equal to +0.17 rating points, and when all brands optimized, per-brand payoff collapsed from +0.802 to +0.007 while non-participating brands received zero recommendations.",
    "date": "2026-06-16",
    "application_area": "Brand Visibility & Mentions",
    "source": "arXiv: Incumbent Advantage: Brand Bias and Cognitive Manipulation Dynamics in LLM Recommendation Systems (arXiv 2606.17443)",
    "source_link": "https://arxiv.org/abs/2606.17443",
    "credibility": "Medium",
    "credibility_rationale": "Academic arXiv preprint by Xi Chu and Yupeng Hou with no venue acceptance or institutional affiliation listed, so not peer reviewed. The controlled-catalog experiment across three commercial LLMs is documented with defined metrics, but it covers a single product domain on the authors' own simulated catalogs and is uncorroborated, so Medium rather than High.",
    "evidence_type": "paper",
    "publisher": "arXiv (Xi Chu, Yupeng Hou)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "61.7% of AI brand appearances are 'ghost citations' where the site is linked but the brand is never named; Gemini names brands in 83.7% of appearances vs 20.7% for ChatGPT.",
    "description": "Semrush (with Kevin Indig) logged 3,981 domain appearances for 115 prompts across 14 countries on ChatGPT, Google AIO, Gemini and AI Mode. 38.3% of appearances included a brand mention, 74.9% a citation, 13.2% both; comparative content earned 2.4x more mentions than informational content.",
    "date": "2026-06-09",
    "application_area": "Brand Visibility & Mentions",
    "source": "Semrush: Why 62% of AI citations don't lead to brand mentions [Study]",
    "source_link": "https://www.semrush.com/blog/the-ghost-citations-study/",
    "credibility": "High",
    "credibility_rationale": "Large established data vendor with documented method (3,981 appearances, 115 prompts, 14 countries, 4 engines), widely re-reported. Modest prompt sample. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Semrush",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "For branded prompts, 57% of 23,387 LLM-cited sources were reviews and social proof, while brand-owned About, Home and FAQ pages made up only about 4%.",
    "description": "Omniscient Digital ran 240 branded prompts (4 industries, 6 intent types) on ChatGPT, Perplexity, Gemini, AI Mode and AIO and classified 23,387 cited sources: reviews/social proof 57%, directories 17%, product/commercial 12%, thought leadership 5.4%, About 1.92%, home 1.82%, FAQ 0.41%.",
    "date": "2026-01-20",
    "application_area": "Brand Visibility & Mentions",
    "source": "Omniscient Digital: Which Content Types LLMs Cite Most: 23,000+ AI Citations Analyzed",
    "source_link": "https://beomniscient.com/blog/content-types-cited-in-llms/",
    "credibility": "Medium",
    "credibility_rationale": "B2B content agency with described method but a small, branded-only prompt sample; repeated only by minor marketing sites.",
    "evidence_type": "study",
    "publisher": "Omniscient Digital",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Across 730,000 response pairs, Google AI Mode and AI Overviews share only 13.7% of citations despite 86% semantic similarity; AI Mode names ~3.3 entities per answer vs 1.3.",
    "description": "Ahrefs compared 730,000 AI Overview / AI Mode response pairs (540,000 for the citation analysis): 89.7% of pairs exceeded 0.8 cosine similarity, but citation overlap was 13.7% (16.3% for top-3 citations), word-level Jaccard overlap 16% and identical opening sentences 2.51%. 59.41% of AIO responses and 34.66% of AI Mode responses contained no brands or entities, AI Mode lacked citations in 3% of responses vs 11% for AIOs, and a brand appearing in an AIO had a 61% chance of also appearing in the AI Mode response.",
    "date": "2025-12-15",
    "application_area": "Brand Visibility & Mentions",
    "source": "Ahrefs: Are AI Mode and AI Overviews Just Different Versions of the Same Answer?",
    "source_link": "https://ahrefs.com/blog/ai-overviews-vs-ai-mode",
    "credibility": "High",
    "credibility_rationale": "First-party Ahrefs study by Despina Gavoyannis using 730,000 AI Overview / AI Mode response pairs with documented measures (cosine semantic similarity, Jaccard word overlap, citation overlap, entity counts); large sample and transparent metrics from a major SEO data vendor.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "YouTube mentions (~0.737) and branded web mentions (0.664) correlate far more strongly with ChatGPT brand visibility than backlinks (~0.2-0.25) or Domain Rating (0.266).",
    "description": "Ahrefs computed Spearman correlations for 75,000 brands (DR > 40) using Brand Radar data across ChatGPT, Google AI Mode and AI Overviews. YouTube mentions were the strongest correlate on every engine (~0.71-0.74), ahead of branded web mentions (0.664/0.709/0.656) and far ahead of backlinks; Domain Rating correlated only 0.266-0.326.",
    "date": "2025-12-12",
    "application_area": "Brand Visibility & Mentions",
    "source": "Ahrefs: AI Brand Visibility Correlations: What Drives Brand Mentions in ChatGPT, AI Mode, and AI Overviews (75,000 Brands)",
    "source_link": "https://ahrefs.com/blog/ai-brand-visibility-correlations",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor; documented sample (75,000 brands, DR > 40, Brand Radar) and method (Spearman); covered by TheNextWeb, Morningstar and Yahoo Finance. Spot-check confirmed 0.737, 0.664 and DR 0.266-0.326.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Brands cited inside a Google AI Overview earned 35% higher organic CTR (0.70% vs 0.52%) and 91% higher paid CTR (7.89% vs 4.14%) than uncited brands on the same queries.",
    "description": "Seer Interactive compared Q3 2025 CTR on AIO queries where the brand was cited vs not cited (3,119 informational queries, 42 client organizations, GSC and Google Ads). Uplift applied to organic and paid, though absolute CTRs stayed far below pre-AIO levels.",
    "date": "2025-11-04",
    "application_area": "Brand Visibility & Mentions",
    "source": "Seer Interactive: AIO Impact on Google CTR: September 2025 Update",
    "source_link": "https://www.seerinteractive.com/insights/aio-impact-on-google-ctr-september-2025-update",
    "credibility": "Medium",
    "credibility_rationale": "Established agency with documented method and trade-press coverage, but a client-portfolio sample rather than an independent large dataset.",
    "evidence_type": "study",
    "publisher": "Seer Interactive",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Across 75,000 brands, branded web mentions correlate 0.664 with Google AI Overview visibility vs 0.218 for backlinks, roughly 3x stronger.",
    "description": "Ahrefs Spearman study (Site Explorer API and Brand Radar) on ~75,000 brands (DR>40, top branded keyword 800+ searches). Web mentions 0.664 > branded anchors 0.527 > branded search volume 0.392 > DR 0.326 > referring domains 0.295 > backlinks 0.218; top mention quartile averaged 169 AIO mentions vs 14; ~26% of brands had zero.",
    "date": "2025-05-26",
    "application_area": "Brand Visibility & Mentions",
    "source": "Ahrefs: An Analysis of AI Overview Brand Visibility Factors (75K Brands Studied)",
    "source_link": "https://ahrefs.com/blog/ai-overview-brand-correlation/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor with proprietary index; documented Spearman method and filters; one of the most referenced Ahrefs GEO studies.",
    "evidence_type": "study",
    "publisher": "Ahrefs (Louise Linehan, data by Xibeijia Guan)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "GPT-4 Turbo ranked products mainly on latent knowledge of product names and was 'minimally influenced by retrieved documents'; all LLMs favored items placed earlier in context.",
    "description": "In the UC Berkeley study's baseline analysis (F-statistics over product name, document content and context position, median across 50 categories x 8 products), GPT-4 Turbo and Llama 3 were heavily influenced by latent knowledge of product names, with GPT-4 Turbo strongly biased toward certain products irrespective of website content, while Llama 3 70B showed greater document sensitivity. Despite a system prompt emphasizing quality-based ranking, all LLMs were significantly influenced by input context position (Mixtral 8x22 median F-statistic ~127), so load order and prior brand knowledge can matter more than on-page copy.",
    "date": "2024-06-05",
    "application_area": "Brand Visibility & Mentions",
    "source": "Pfrommer et al. (UC Berkeley, EMNLP 2024): Ranking Manipulation for Conversational Search Engines",
    "source_link": "https://arxiv.org/abs/2406.03589",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed EMNLP 2024 Main-track paper from UC Berkeley EECS (Pfrommer, Bai, Gautam, Sojoudi); results are backed by a released dataset and the F-statistic analysis reported in Section 5.1 and Figure 2d.",
    "evidence_type": "paper",
    "publisher": "arXiv / EMNLP 2024 (UC Berkeley EECS)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "AI-referred visits to U.S. retail sites converted 60% better than non-AI traffic in July 2026, the 11th straight month, with AI referrals up 62% YoY and 1,219% vs October 2024.",
    "description": "Adobe Analytics data for July 2026 (1T+ visits to U.S. retail sites, 200+ of the top 2,000 retailers) as reported by Digital Commerce 360 (Aug 19, 2026). AI-referred shoppers generated 53% more revenue per visit, 59% more time on site, 33% lower bounce rate and 28% more add-to-carts.",
    "date": "2026-08-19",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Adobe Analytics (reported by Digital Commerce 360): Adobe: AI-referral traffic spending, converting more than counterparts",
    "source_link": "https://www.digitalcommerce360.com/2026/08/19/adobe-ai-referral-traffic-data-july-2026/",
    "credibility": "Medium",
    "credibility_rationale": "Underlying data is Adobe Analytics with stated methodology and the series is corroborated by earlier PYMNTS/TechCrunch reports, but the row links secondary reporting without a direct Adobe page or named spokesperson. Spot-check confirmed figures.",
    "evidence_type": "statistic",
    "publisher": "Adobe Analytics (reported by Digital Commerce 360, Abbas Haleem)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "ChatGPT citation presence in US prompts rose from 1.6% to 6.8% (June 2025-May 2026), and homepage referrals jumped from ~26-29% to ~62-63% after the May 7, 2026 brand-link update.",
    "description": "Similarweb clickstream data (2026 Generative AI Landscape report) tracked citation presence and referral composition around ChatGPT's May 7, 2026 clickable-brand-link update. ChatGPT's share of generative AI traffic fell from ~76% to ~53% while Gemini rose to ~27-28% and Claude ~9%; average monthly generative AI visits grew 70% YoY to 9.5B.",
    "date": "2026-07-29",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Similarweb: AI Search Stats in 2026: Market Share, Referral, and Citation",
    "source_link": "https://aisearch.similarweb.com/blog/gen-ai-stats/",
    "credibility": "High",
    "credibility_rationale": "Major web-traffic data vendor with a large clickstream panel; figures from its own landscape report and weekly referral data, routinely covered by TechCrunch and re-cited by third parties. Spot-check confirmed all figures.",
    "evidence_type": "statistic",
    "publisher": "Similarweb",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "ChatGPT drives 92.4% of trackable LLM referral traffic and monthly LLM sessions grew 9.9x (Nov 2024 to May 2026), yet LLM traffic remains only about 1-2% of sessions.",
    "description": "Previsible's third AI Traffic Study analyzed 6.77M LLM-driven sessions across 166 GA4 properties in ten industries over 19 months; monthly sessions rose from 65,249 to 644,478. ChatGPT grew 12.8x, Claude 64x, Gemini 3.2x; Perplexity fell 61% and Copilot 96% from peaks. Peak penetration 1.71% (SMB).",
    "date": "2026-07-06",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Previsible: 2026 AI Traffic Report - ChatGPT Wins 92% Share",
    "source_link": "https://previsible.com/seo-strategy/ai-traffic-report-july-2026/",
    "credibility": "Medium",
    "credibility_rationale": "SEO/GEO agency with experienced leadership and a documented recurring method (6.77M sessions, 166 GA4 properties), but a convenience sample of client properties. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "Previsible",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Weekly AI chatbot news use rose from 7% to 10% globally in a year, yet only 4% always or often click through to underlying news sources from AI (19% from search).",
    "description": "The Reuters Institute Digital News Report 2026 (48 markets, published 16 June 2026) found chatbot news use reached 17% of the youngest age group, with follow-up questions the top use case (42% of users), then latest news (35%), summaries (34%) and assessing source reliability (33%). Trust in news from AI chatbots is 20% among the general population (44% among chatbot users); growth came largely from Asia, Africa, Latin America and Southern and Eastern Europe, while the USA, UK, France and Germany showed no increase.",
    "date": "2026-06-16",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Reuters Institute for the Study of Journalism: Digital News Report 2026 - Emerging uses of AI chatbots for news and what it means for journalism",
    "source_link": "https://reutersinstitute.politics.ox.ac.uk/digital-news-report/2026/emerging-uses-ai-chatbots-news-and-what-it-means-journalism",
    "credibility": "High",
    "credibility_rationale": "University of Oxford research institute; the Digital News Report is an annual YouGov-fielded survey across 48 markets (roughly 2,000 respondents per market) with a fully published methodology, and is the most widely cited source on digital news consumption globally.",
    "evidence_type": "study",
    "publisher": "Reuters Institute for the Study of Journalism, University of Oxford",
    "publisher_type": "research-institution"
  },
  {
    "finding": "68.01% of US Google searches ended without a click in Jan-Apr 2026 (60.45% in 2024 on the same panel); AI Mode was 0.34% of searches and AI tools sent under 1% of traffic.",
    "description": "SparkToro's 2026 zero-click update used Similarweb's US desktop and mobile clickstream panel for January-April 2026. Clicks to any destination fell 9.51 points (a 22.9% decline) versus 2024 while the share of searchers performing an additional search rose 7.2 points; the 60.45% 2024 baseline is from the Similarweb panel, not the 58.5% Datos figure from the 2024 study.",
    "date": "2026-06-09",
    "application_area": "Traffic, Leads & Conversion",
    "source": "SparkToro (Rand Fishkin) with Similarweb: In 2026, Less than One Third of Google Searches Still Send a Click",
    "source_link": "https://sparktoro.com/blog/in-2026-less-than-one-third-of-google-searches-still-send-a-click/",
    "credibility": "High",
    "credibility_rationale": "Based on Similarweb's US desktop and mobile clickstream panel for January-April 2026, with methodology and assumptions stated in the post; authored by Rand Fishkin and covered by Search Engine Land, Search Engine Roundtable and Similarweb's own blog. Similarweb is a large data vendor.",
    "evidence_type": "study",
    "publisher": "SparkToro (with Similarweb)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Being cited in a Google AI Overview delivered 120% more organic clicks per impression in 2025 (2.07% CTR when cited vs 0.94% when not; 3.35% when no AIO shown).",
    "description": "Seer Interactive's 2026 update (McDonald, Cooley, Williams) analyzed 53 brands, 5.47M queries, 2.43B organic and 296.9M paid impressions, Jan 2025-Feb 2026. AIOs appeared on 36% of informational, 8% of commercial and 5% of transactional queries; cited CTR still trailed no-AIO by 38%.",
    "date": "2026-04-24",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Seer Interactive: AIO Impact on Google CTR: 2026 Update",
    "source_link": "https://www.seerinteractive.com/insights/aio-impact-on-google-ctr-2026-update",
    "credibility": "Medium",
    "credibility_rationale": "Established agency with a large, well-documented client-portfolio sample and SEL/Inc./Fast Company coverage; rated Medium for consistency with other agency client-data studies. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Seer Interactive",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Search referrals fell 60% for small, 47% for mid-sized and 22% for large publishers over two years, while AI chatbots still drive under 1% of publisher page-view referrals.",
    "description": "Chartbeat data across thousands of news sites, reported by Axios (Mar 17, 2026) and summarized by Search Engine Journal. Google Search page views fell 34% and Discover 15% between Dec 2024 and Dec 2025; ChatGPT referrals grew 200%+ but chatbots remain under 1% of referrals. Tertiary reporting.",
    "date": "2026-03-17",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Search Engine Journal (Chartbeat data via Axios): Search Referral Traffic Down 60% For Small Publishers, Data Shows",
    "source_link": "https://www.searchenginejournal.com/search-referral-traffic-down-60-for-small-publishers-data-shows/569959/",
    "credibility": "Medium",
    "credibility_rationale": "Chartbeat is a major publisher-analytics vendor and Axios/SEJ are reputable, but this is tertiary reporting with no published Chartbeat methodology and the Axios primary could not be fetched.",
    "evidence_type": "statistic",
    "publisher": "Search Engine Journal (reporting Chartbeat data via Axios)",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "ChatGPT referrals to publishers grew 200%+ YoY but AI chatbots stay under 1% of pageviews; AI-referred readers engage most with evergreen content (10.6 vs 4.8 pages).",
    "description": "Chartbeat's Q1 2026 report used network data from January 1, 2024 to February 9, 2026. AI-referred readers viewed 10.6 pages per article in Home & Garden, 8.6 in Health and 8.5 in Science & Education versus 4.8 for News & Media, which receives the highest total AI-referred pageviews but the lowest engagement; by article category, Economy, Business & Finance led with 7.35M AI-referred pageviews, followed by Science & Technology (1.97M) and Environment (1.92M).",
    "date": "2026-03",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Chartbeat: Navigating the New Traffic Landscape: What's Really Happening with Search, Dark Social, and AI",
    "source_link": "https://lp.chartbeat.com/hubfs/Chartbeat-Report-2026Q1.pdf",
    "credibility": "High",
    "credibility_rationale": "Chartbeat is the leading real-time analytics platform for publishers (4,000+ sites, more than 50 billion monthly pageviews across 70+ countries). The report states its data window (January 1, 2024 to February 9, 2026), gives absolute figures and defines its tiers, and was covered by trade press. Limitations: the PDF carries no exact publication date and it is a vendor report without full methodology, but sample size and documentation meet the large-data-vendor bar.",
    "evidence_type": "study",
    "publisher": "Chartbeat, Inc.",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "ChatGPT referrals to 94 ecommerce stores converted at 1.81% vs 1.39% for non-branded organic search (31% higher), with 10.3% higher revenue per session.",
    "description": "180 Marketing (Jeff Oxford) analyzed 12 months of GA4 data (2025) from 94 ecommerce stores, excluding homepage and blog traffic. ChatGPT AOV was 14.3% lower ($204 vs $238); ChatGPT drove 135K sessions and $474K revenue vs 9.46M sessions and $32.1M for non-branded organic, growing 1,079% over the year.",
    "date": "2026-02-23",
    "application_area": "Traffic, Leads & Conversion",
    "source": "180 Marketing: ChatGPT Traffic Converts 31% Better than Non-Branded Organic Search (94 eCommerce Sites)",
    "source_link": "https://www.180marketing.com/chatgpt-vs-organic-search-conversion-rates/",
    "credibility": "Medium",
    "credibility_rationale": "Ecommerce SEO agency with described method on client data; modest ChatGPT sample (135K sessions); no independent citations verified.",
    "evidence_type": "study",
    "publisher": "180 Marketing (Jeff Oxford)",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "By December 2025 an AI Overview correlated with a 58% lower CTR for the top-ranking page (up from 34.5% in April 2025), with position two down 50.8% and position ten down 19.4%.",
    "description": "Ahrefs repeated its 300,000-keyword aggregated GSC desktop method comparing December 2023 with December 2025. Position-1 CTR on AIO keywords fell from 0.073 to 0.016 vs a projected 0.037; non-AIO informational keywords fell from 0.076 to 0.039.",
    "date": "2026-02-04",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Ahrefs: AI Overviews Reduce Clicks by 58% (December 2025 Update)",
    "source_link": "https://ahrefs.com/blog/ai-overviews-reduce-clicks-update",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor repeating a documented 300,000-keyword method; covered by Search Engine Land, PPC Land and Business Wire. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "AI-app click-through rates to sites without an AI licensing deal fell nearly 3x in 2025 (0.8% Q2 to 0.27% Q4), and sites with deals fell over 6.5x (8.8% Q1 to 1.33% Q4).",
    "description": "TollBit's 'State of the Bots' Q3-Q4 2025 report ('The Leaky Pipes'), based on traffic across its publisher network (CTR figures from a cohort of sites onboarded before Q2 2025, ~80.5% of whose referrals came from Google), found AI apps averaged just 0.12% of site referrals and aggregate AI-app CTR remained over 13x lower than pre-AI Google SERPs. TollBit says 1:1 deals are not necessarily insulating publishers from low AI referrals; human traffic rose 1% from Q2 to Q3 and fell 5% from Q3 to Q4 2025.",
    "date": "2026-02-04",
    "application_area": "Traffic, Leads & Conversion",
    "source": "TollBit: State of the Bots, 2025 Q3 & Q4 - The Leaky Pipes",
    "source_link": "https://tollbit.com/state-of-the-bots/q3-q4-2025/",
    "credibility": "Medium",
    "credibility_rationale": "First-party, methodology-described study by an AI-bot monetization vendor with a large publisher network (6,800+ publishers per TollBit), widely covered by The Register, Digiday and Press Gazette. Rated Medium because TollBit has a commercial stake in the narrative (it sells bot paywalls and licensed RAG), its network skews to news publishers, and it acknowledges its bot-vs-human detection is conservative.",
    "evidence_type": "study",
    "publisher": "TollBit",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "AI referral traffic is just 1.08% of website traffic across 10 industries, 87.4% of it from ChatGPT, while AI Overviews appear on 25.11% of Google searches.",
    "description": "Conductor's 2026 AEO/GEO Benchmarks Report: 13,770 domains across 10 industries, 3.5M prompts (May-Sept 2025), 17M AI responses, 100M citations, 21.9M Google searches and 3.3B+ sessions from 1,215 enterprise domains. AIOs appeared on 5.5M of 21.9M searches, from 48.75% (Healthcare) to 4.48% (Real Estate); AI referrals highest in IT at 2.80%.",
    "date": "2025-11-13",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Conductor: The 2026 AEO / GEO Benchmarks Report",
    "source_link": "https://www.conductor.com/academy/aeo-geo-benchmarks-report/",
    "credibility": "High",
    "credibility_rationale": "Enterprise SEO/AEO platform with a large, documented sample; launch covered via BusinessWire (Nov 13, 2025), Yahoo Finance and MarTech Cube; figures widely re-cited. Spot-check confirmed numbers; page shows 'last updated July 6, 2026' but first release was Nov 13, 2025.",
    "evidence_type": "study",
    "publisher": "Conductor",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "LLM-referred visitors to publisher sites signed up at 1.66% vs 0.15% for search and subscribed at 1.34% vs 0.55%, while AI referrals grew 155.6% in eight months yet stayed under 1% of traffic.",
    "description": "Microsoft Clarity analyzed 1,277 domains (1,200+ publisher/news sites) over eight months with one month of conversion data; search grew 24.0%, social 21.5%, direct 14.9%. 52% of domains converted AI visitors within the month; Copilot had the highest subscription rate (17x direct, 15x search).",
    "date": "2025-11-06",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Microsoft Clarity: AI Traffic Converts at 3x the Rate of Other Channels (Study)",
    "source_link": "https://clarity.microsoft.com/blog/ai-traffic-converts-at-3x-the-rate-of-other-channels-study/",
    "credibility": "High",
    "credibility_rationale": "Published by Microsoft using first-party Clarity analytics with documented sample and period; re-reported by PPC Land and Digiday. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Microsoft Clarity",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Organic CTR on informational queries with an AI Overview fell 61% (1.76% to 0.61%) and paid CTR fell 68% (19.70% to 6.34%) between June 2024 and September 2025.",
    "description": "Seer Interactive analyzed 3,119 informational queries across 42 client organizations (25.1M organic, 1.1M paid impressions) using GSC and Google Ads data with AIO presence from SeerSignals and ZipTie. Non-AIO queries also declined 41% (2.74% to 1.62%).",
    "date": "2025-11-04",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Seer Interactive: AIO Impact on Google CTR: September 2025 Update",
    "source_link": "https://www.seerinteractive.com/insights/aio-impact-on-google-ctr-september-2025-update",
    "credibility": "Medium",
    "credibility_rationale": "Established agency with documented method (Tracy McDonald; 3,119 queries, 42 orgs, sources named) and coverage by Search Engine Land and Inc., but a client-portfolio sample rather than an independent large dataset.",
    "evidence_type": "study",
    "publisher": "Seer Interactive",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Weekly generative AI use in six countries rose from 18% to 34% in a year; 54% saw AI search answers last week, of whom 33% often click through to sources and 28% rarely do.",
    "description": "The Reuters Institute Generative AI and News Report 2025 surveyed about 2,000 people each in Argentina, Denmark, France, Japan, the UK and the US via YouGov in June-July 2025. Information-seeking became the lead use case (11% to 24% weekly), ChatGPT was used weekly by 22%, weekly use of AI for news doubled from 3% to 6%, and 50% of those who saw AI search answers said they trust them, with exposure ranging from 70% in Argentina to 29% in France.",
    "date": "2025-10-07",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Reuters Institute for the Study of Journalism: Generative AI and News Report 2025 - How people think about AI's role in journalism and society",
    "source_link": "https://reutersinstitute.politics.ox.ac.uk/generative-ai-and-news-report-2025-how-people-think-about-ais-role-journalism-and-society",
    "credibility": "High",
    "credibility_rationale": "University of Oxford research institute; YouGov online survey of roughly 2,000 respondents in each of six countries, fieldwork 5 June to 15 July 2025, quota-sampled and weighted to census data, with methodology published in the report.",
    "evidence_type": "study",
    "publisher": "Reuters Institute for the Study of Journalism, University of Oxford",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Digital Content Next members saw a median 10% YoY decline in Google Search referrals over eight weeks in May-June 2025 (news -7%, non-news -14%), with sites losing 1% to 25%.",
    "description": "DCN surveyed 19 of ~40 member companies (12 news, 7 non-news) on self-reported Google referral data over eight weeks; median weekly referrals were down almost every week. DCN CEO Jason Kint attributed losses to AI Overviews. Digiday cited because DCN's post returned 404.",
    "date": "2025-08-15",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Digiday (Digital Content Next survey): Google AI Overviews linked to 25% drop in publisher referral traffic, new data shows",
    "source_link": "https://digiday.com/media/google-ai-overviews-linked-to-25-drop-in-publisher-referral-traffic-new-data-shows/",
    "credibility": "Medium",
    "credibility_rationale": "Established trade association and reputable trade outlet with method described, but a small, self-reported sample with attribution based on interpretation rather than controlled analysis.",
    "evidence_type": "study",
    "publisher": "Digiday (reporting a Digital Content Next member survey)",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "Google says total organic click volume from Search to websites is 'relatively stable year-over-year' and it is sending 'slightly more quality clicks' than a year ago.",
    "description": "Liz Reid (Aug 6, 2025) responded to AI Overview/AI Mode traffic concerns: total organic click volume 'relatively stable year-over-year' and 'average click quality has increased' (clicks where users don't quickly click back). No numbers or charts; criticized by publishers (Press Gazette, DCN) and contradicted by per-query CTR studies.",
    "date": "2025-08-06",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Google (The Keyword): AI in Search is driving more queries and higher quality clicks",
    "source_link": "https://blog.google/products-and-platforms/products/search/ai-search-driving-more-queries-higher-quality-clicks/",
    "credibility": "High",
    "credibility_rationale": "First-party statement by Google's Head of Search; widely covered by Engadget, Forbes, SEJ and Press Gazette; authoritative as a platform position, not independent evidence.",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "When a Google AI Overview appeared for UK news queries, publisher CTR fell 47.5% on desktop and 37.7% on mobile, and top AIO citation position only marginally improved clicks.",
    "description": "Authoritas measured 3,500 UK news-related search terms (April 16-22, 2025); AIOs appeared on ~12.2% (rising to ~50% for health and horoscope queries). Even top AIO-position sites saw only minor click gains. Findings filed with the UK CMA; cited via Press Gazette as the Authoritas primary page could not be located.",
    "date": "2025-07-30",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Press Gazette (Authoritas data): Publishers lose 50% of clickthrough rate due to Google AI Overviews",
    "source_link": "https://pressgazette.co.uk/media-audience-and-business-data/google-ai-overviews-publishers-report-clickthroughs-authoritas-report/",
    "credibility": "Medium",
    "credibility_rationale": "Established trade outlet documenting the Authoritas method, independently reported by The Media Leader; but a single-week sample, primary report not located, and Authoritas is a commercial vendor.",
    "evidence_type": "study",
    "publisher": "Press Gazette (reporting Authoritas research)",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "When a Google AI summary appears, users click a traditional result on 8% of visits vs 15% without one, click a summary link 1% of the time, and end the session 26% vs 16%.",
    "description": "Pew Research Center analyzed 68,879 unique Google searches by 900 U.S. adults (Ipsos KnowledgePanel Digital, March 2025); 12,593 (18%) produced an AI summary. 88% of summaries cited three or more sources.",
    "date": "2025-07-22",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Pew Research Center: Google users are less likely to click on links when an AI summary appears in the results",
    "source_link": "https://www.pewresearch.org/short-reads/2025/07/22/google-users-are-less-likely-to-click-on-links-when-an-ai-summary-appears-in-the-results/",
    "credibility": "High",
    "credibility_rationale": "Nonpartisan research institution with documented panel method; figures widely corroborated by SEL, SEJ, Statista and others. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Pew Research Center",
    "publisher_type": "research-institution"
  },
  {
    "finding": "The average AI search visitor is 4.4x as valuable as a traditional organic visitor by conversion rate, and pages ChatGPT cites rank at position 21+ almost 90% of the time.",
    "description": "Semrush analyzed 500+ digital-marketing/SEO topics as search terms and prompts across Google AIO, AI Mode, ChatGPT, Claude and Perplexity, projecting AI search may out-refer traditional search for those topics by early 2028. The conversion rates behind the 4.4x figure are not disclosed and Semrush labels projections as extrapolations.",
    "date": "2025-07-21",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Semrush: We Studied the Impact of AI Search on SEO Traffic",
    "source_link": "https://www.semrush.com/blog/ai-search-seo-traffic-study/",
    "credibility": "Medium",
    "credibility_rationale": "Semrush is a major data vendor and the study (Rachel Handley) was covered by MarTech and PPC Land, but spot-check confirmed the headline 4.4x figure is asserted without any underlying conversion rates or traffic dataset, so the row's specific claim is not methodology-documented.",
    "evidence_type": "study",
    "publisher": "Semrush",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Cloudflare CEO: versus the original Google model, earning the same traffic is ~10x harder from Google, 750x harder from OpenAI and 30,000x harder from Anthropic.",
    "description": "Matthew Prince's Content Independence Day post (July 1, 2025) announcing the default block on AI crawlers unless they pay and the pay-per-crawl launch; multipliers derive from Cloudflare's network-wide crawl-to-visit comparisons.",
    "date": "2025-07-01",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Cloudflare Blog: Content Independence Day: no AI crawl without compensation!",
    "source_link": "https://blog.cloudflare.com/content-independence-day-no-ai-crawl-without-compensation/",
    "credibility": "High",
    "credibility_rationale": "Authored by Cloudflare's CEO on the official blog drawing on network-wide data; the announcement was covered by major outlets and is widely cited.",
    "evidence_type": "statistic",
    "publisher": "Cloudflare (Matthew Prince)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Organic search traffic to news sites fell 26% after Google AI Overviews launched, while ChatGPT referrals to publishers grew 25x and ChatGPT news queries rose 212% in 18 months.",
    "description": "Similarweb clickstream report on generative AI and U.S. publishers (data through May 2025). Per TechCrunch's coverage of the same report: organic news visits fell from over 2.3B to under 1.7B, zero-click news searches rose from 56% to nearly 69%, and ChatGPT referrals rose from under 1M (Jan-May 2024) to over 25M in 2025; Reuters and NY Post topped referral recipients.",
    "date": "2025-07",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Similarweb: GenAI and How It's Impacting US Publishers",
    "source_link": "https://www.similarweb.com/corp/reports/generative-ai-publishers/",
    "credibility": "High",
    "credibility_rationale": "Major panel-based clickstream data vendor; headline figures on the landing page, corroborated by TechCrunch and Press Gazette coverage.",
    "evidence_type": "study",
    "publisher": "Similarweb",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Google states that clicks to websites from results pages with AI Overviews are higher quality, with users more likely to spend more time on the site.",
    "description": "Same May 21, 2025 post: 'when people click to a website from search results pages with AI Overviews, these clicks are higher quality, where users are more likely to spend more time on the site.' No quantitative data given; Google also asserts people search more often and ask more complex questions.",
    "date": "2025-05-21",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Google Search Central Blog: Top ways to ensure your content performs well in Google's AI experiences on Search",
    "source_link": "https://developers.google.com/search/blog/2025/05/succeeding-in-ai-search",
    "credibility": "High",
    "credibility_rationale": "Official platform statement by John Mueller; authoritative as Google's position but backed by no published data.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central (John Mueller)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "One year after AI Overviews launched, total search impressions rose 49% while CTR fell nearly 30%; AIOs appeared on over 11% of Google queries (+22% since debut).",
    "description": "BrightEdge press release marking one year of AI Overviews (May 14, 2025). Longer, complex queries in AIOs grew 49% while ranking-style and comparison queries decreased 60% and 14%; Healthcare, Education, B2B Tech and Insurance led AIO presence, and Google retained over 90% search share.",
    "date": "2025-05-14",
    "application_area": "Traffic, Leads & Conversion",
    "source": "BrightEdge: One Year Into Google AI Overviews, BrightEdge Data Reveals Google Search Usage",
    "source_link": "https://www.brightedge.com/news/press-releases/one-year-google-ai-overviews-brightedge-data-reveals-google-search-usage",
    "credibility": "Medium",
    "credibility_rationale": "First-party research from BrightEdge, a large enterprise SEO data vendor, widely re-reported by Search Engine Land and Search Engine Journal, but delivered as a press release that does not document the sample, query set or measurement method behind the impression and CTR figures, so Medium rather than High.",
    "evidence_type": "statistic",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "The presence of a Google AI Overview correlated with a 34.5% lower average CTR for the top-ranking page (March 2024 vs March 2025).",
    "description": "Ahrefs compared aggregated Google Search Console desktop CTR for 300,000 keywords (150K with AIOs, 150K informational without) between March 2024 and March 2025. Position-1 CTR on AIO keywords fell from 0.073 to 0.026 vs a 0.040 forecast; non-AIO informational keywords fell from 0.056 to 0.031.",
    "date": "2025-04-17",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Ahrefs: AI Overviews Reduce Clicks by 34.5%",
    "source_link": "https://ahrefs.com/blog/ai-overviews-reduce-clicks/",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor; 300,000-keyword sample with documented GSC-based method (Ryan Law, data scientist Xibeijia Guan); covered by Search Engine Land and eMarketer and widely re-cited.",
    "evidence_type": "study",
    "publisher": "Ahrefs",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Keywords triggering AI Overviews lost 15.49% CTR on average (non-branded -19.98%, AIO plus Featured Snippet -37.04%), while branded AIO keywords gained 18.68%.",
    "description": "Amsive studied 700,000 keywords across 10 client websites in five industries, focusing on 10,000 AIO-triggering keywords that already ranked. Only 4.79% of branded keywords triggered an AIO; keywords outside the top 3 dropped 27.04%.",
    "date": "2025-04-16",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Amsive: Google AI Overviews: New CTR Study Reveals How to Navigate Negative SERP Impact",
    "source_link": "https://www.amsive.com/insights/seo/google-ai-overviews-new-research-reveals-how-to-navigate-click-drop-off/",
    "credibility": "Medium",
    "credibility_rationale": "Mid-size agency study (Will Guevara) with sample described and Search Engine Land coverage, but only 10 websites and an incompletely specified baseline period.",
    "evidence_type": "study",
    "publisher": "Amsive",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Bain: 80% of consumers rely on AI-written search results for at least 40% of searches, 60% of searches end without a click-through, and organic traffic is down an estimated 15-25%.",
    "description": "Bain & Company consumer research released Feb 19, 2025 also found 68% of LLM users rely on chatbots for research. The release discloses no sample size, method or fieldwork dates; PR Newswire distribution cited because bain.com blocked automated fetch.",
    "date": "2025-02-19",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Bain & Company: Consumer reliance on AI search results signals new era of marketing (press release via PR Newswire)",
    "source_link": "https://www.prnewswire.com/news-releases/consumer-reliance-on-ai-search-results-signals-new-era-of-marketing--bain--company-302379679.html",
    "credibility": "Medium",
    "credibility_rationale": "Bain is an authoritative consultancy and all four figures are stated verbatim on its official release, but no sample size, method or dates are disclosed and the 15-25% traffic figure is an estimate; spot-check confirmed the figures and the absence of methodology.",
    "evidence_type": "study",
    "publisher": "Bain & Company",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "58.5% of US and 59.7% of EU Google searches ended without a click in 2024; only 360 (US) and 374 (EU) of every 1,000 searches sent a click to the open web.",
    "description": "SparkToro/Datos clickstream study using Datos US and EU panels from September 2022 to May 2024. Of the remaining clicks, almost 30% went to Google-owned properties and about 1% of all clicks went to paid ads; zero-click searchers either ended the session (~37%) or re-searched (~22%).",
    "date": "2024-07-02",
    "application_area": "Traffic, Leads & Conversion",
    "source": "SparkToro (Rand Fishkin) with Datos: 2024 Zero-Click Search Study",
    "source_link": "https://sparktoro.com/blog/2024-zero-click-search-study-for-every-1000-us-google-searches-only-374-clicks-go-to-the-open-web-in-the-eu-its-360/",
    "credibility": "High",
    "credibility_rationale": "Large-sample clickstream study using Datos (Semrush) US and EU panel data from September 2022 to May 2024, with methodology, definitions and caveats documented, authored by Rand Fishkin and widely cited by Search Engine Land and industry press. Caveat: the article URL and title reverse the US and EU open-web click figures; the body text (US 360, EU 374) is authoritative.",
    "evidence_type": "study",
    "publisher": "SparkToro (with Datos, a Semrush company)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Google claims links included in AI Overviews get more clicks than if the page had appeared as a traditional web listing for that query.",
    "description": "Google I/O 2024 post by Elizabeth Reid (May 14, 2024): 'the links included in AI Overviews get more clicks than if the page had appeared as a traditional web listing for that query,' and people visit a greater diversity of websites. No supporting data; widely contested by third-party CTR studies.",
    "date": "2024-05-14",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Google (The Keyword): Generative AI in Search: Let Google do the searching for you",
    "source_link": "https://blog.google/products-and-platforms/products/search/generative-ai-google-search-may-2024/",
    "credibility": "High",
    "credibility_rationale": "First-party statement by Google's VP of Search on its official blog; authoritative as a platform claim, not as evidence of click outcomes (disputed by Press Gazette, SEL, Fortune coverage).",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Google.com sent 63.41% of US web referrals across the top 170 sites (Jan 2023-Jan 2024), nearly 10x the next referrer; OpenAI/ChatGPT was 0.21% despite growing 82%.",
    "description": "Datos and SparkToro analysis of 13 months of US clickstream data across the top 170 referring domains. Microsoft properties combined were 7.21% of referrals and Reddit referrals fell 30% over the period, establishing that AI tools were a negligible referral source before AI Overviews launched.",
    "date": "2024-03-11",
    "application_area": "Traffic, Leads & Conversion",
    "source": "SparkToro (Rand Fishkin) with Datos: Who Sends Traffic on the Web and How Much?",
    "source_link": "https://sparktoro.com/blog/who-sends-traffic-on-the-web-and-how-much-new-research-from-datos-sparktoro/",
    "credibility": "High",
    "credibility_rationale": "13-month US clickstream analysis (January 2023 to January 2024) from Datos (Semrush) across the top 170 referring domains, with methodology described and a follow-up webinar with Datos' CEO; authored by Rand Fishkin and widely cited in industry press. Caveat: panel-based data reflecting the pre-AI-Overviews period.",
    "evidence_type": "study",
    "publisher": "SparkToro (with Datos, a Semrush company)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Gartner forecast that traditional search engine volume will fall 25% by 2026 as generative AI chatbots become substitute answer engines.",
    "description": "Gartner forecast (Feb 19, 2024) reported by MediaPost, quoting VP Analyst Alan Antin that GenAI tools will replace queries previously executed in traditional search engines. A forecast, not a measurement; the source does not assess whether it materialized.",
    "date": "2024-02-19",
    "application_area": "Traffic, Leads & Conversion",
    "source": "Gartner (reported by MediaPost): Traditional Search Forecast To Fall 25% By 2026: Gartner",
    "source_link": "https://www.mediapost.com/publications/article/393629/traditional-search-forecast-to-fall-25-by-2026-g",
    "credibility": "Medium",
    "credibility_rationale": "Gartner is an authoritative research firm, but the row links a secondary trade-press report (Gartner's release blocked fetch) and the claim is a forecast rather than a measured result. Spot-check confirmed the page and quote.",
    "evidence_type": "analysis",
    "publisher": "Gartner (reported by MediaPost, Laurie Sullivan)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "9.3% of the world's top 1,000 domains (93 sites) serve an llms.txt file as of September 18, 2026, up from 0.3% in June 2025.",
    "description": "Rankability's adoption tracker requests /llms.txt and /llms-full.txt on the top 1,000 and 10,000 domains, counting only HTTP 200 text responses with Markdown structure. Also 8.3% (830/10,000) in the top 10,000; the authors state llms.txt confers no ranking, citation or training-permission benefit.",
    "date": "2026-09-18",
    "application_area": "Technical Access & Crawling",
    "source": "Rankability: LLMs.txt Adoption Statistics (llms.txt Adoption Tracker)",
    "source_link": "https://www.rankability.com/data/llms-txt-adoption/",
    "credibility": "Medium",
    "credibility_rationale": "SEO SaaS (Nathan Gotch, Simon L. Smith) with a documented scan method and explicit disclaimers, but a single vendor's scan without third-party corroboration. Spot-check confirmed 9.3%/93, 8.3% and the 0.3% baseline.",
    "evidence_type": "statistic",
    "publisher": "Rankability",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Google says the Google-Extended token only controls Gemini training and grounding; it does not affect Search inclusion, ranking, or AI Overviews/AI Mode eligibility.",
    "description": "Google crawler documentation (updated 2026-07-14) defines Google-Extended as a standalone token for Gemini training and grounding: 'Google-Extended does not impact a site's inclusion in Google Search nor is it used as a ranking signal in Google Search.' AI features documentation says Googlebot robots.txt and snippet controls govern AI features.",
    "date": "2026-07-14",
    "application_area": "Technical Access & Crawling",
    "source": "Google Search Central: Google's common crawlers (Google-Extended)",
    "source_link": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "credibility": "High",
    "credibility_rationale": "Official Google developer documentation; sentence quoted verbatim by Search Engine Journal and consistent with Google's AI features documentation.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "From September 15, 2026, new Cloudflare domains block Training and Agent AI bots by default on pages that display ads, while Search crawlers remain allowed.",
    "description": "Cloudflare's 2026 Content Independence Day post updates the July 2025 one-click block with three AI traffic categories (Search, Agent, Training), reasoning that an ad signals a page meant for humans. Owners can opt out before Sept 15; default for ad-free pages is not explicitly stated.",
    "date": "2026-07-01",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: Your site, your rules: new AI traffic options for all customers",
    "source_link": "https://blog.cloudflare.com/content-independence-day-ai-options/",
    "credibility": "High",
    "credibility_rationale": "Official Cloudflare announcement of its own bot-management defaults, following the widely covered July 2025 policy.",
    "evidence_type": "official-guidance",
    "publisher": "Cloudflare",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Pages that time out for AI crawlers more than 75% of the time received roughly 18x fewer citation events than stable pages, in a 700,000-page Profound sample.",
    "description": "Profound analysis (Jasman Singh, Josh Blyskal) of 700,000 pages over a multi-day period in April 2026, reported second-hand by iPullRank's Mike King (May 8, 2026). HTTP 499 client-disconnect responses proxied retrieval failure for GPTBot, OAI-SearchBot, ClaudeBot and Perplexity agents; framed as exclusion from the candidate set. No standalone Profound report or methodology is published.",
    "date": "2026-05-08",
    "application_area": "Technical Access & Crawling",
    "source": "iPullRank (citing Profound): Quick Tip: How Page Speed Impacts ChatGPT and Perplexity Visibility",
    "source_link": "https://ipullrank.com/page-speed-impacts",
    "credibility": "Medium",
    "credibility_rationale": "Widely cited technical SEO agency relaying data credited to a major AI-visibility vendor with a stated sample, but reported second-hand with no published methodology, controls or independent corroboration. Spot-check confirmed the article's claims and attribution.",
    "evidence_type": "analysis",
    "publisher": "iPullRank (Mike King), citing Profound analysis",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Sites blocking the Google-Extended crawler were never cited by Gemini and were cited less by AI Overviews: 21 popular publishers retrieved by Google Search got zero Gemini citations.",
    "description": "Section 4.3.2 of the NJIT SIGIR 2026 study: 21 publishers retrieved for 20+ queries by Google Search and AIOs (incl. NYTimes, ESPN, CNN, BBC) were never cited by Gemini and all block Google-Extended; many also had fewer AIO citations, despite Google stating Google-Extended does not affect Search or AIO. Observational, no formal test.",
    "date": "2026-04-30",
    "application_area": "Technical Access & Crawling",
    "source": "arXiv / ACM SIGIR 2026 (NJIT): How Generative AI Disrupts Search: An Empirical Study of Google Search, Gemini, and AI Overviews",
    "source_link": "https://arxiv.org/abs/2604.27790",
    "credibility": "High",
    "credibility_rationale": "Same peer-reviewed ACM SIGIR 2026 paper; Section 4.3.2 and Table 3 described explicitly in the full text, though observational without a statistical test.",
    "evidence_type": "paper",
    "publisher": "Riley Grossman et al. (New Jersey Institute of Technology)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "AI agent requests reached 88% of human organic search activity, ~95% of agent traffic came from OpenAI, yet only 19% of sites have directives for ChatGPT-related bots.",
    "description": "BrightEdge press release (April 8, 2026) reports AI agent requests account for approximately 15% of total website traffic and projects agent activity will surpass human-driven search by end of 2026. 81% of companies treat AI agents like traditional bots with obsolete or conflicting rules, 77% focus on blocking training agents while only 38% address user-facing agents and 21% address search agents.",
    "date": "2026-04-08",
    "application_area": "Technical Access & Crawling",
    "source": "BrightEdge: AI Search is Reaching a Tipping Point (press release)",
    "source_link": "https://www.brightedge.com/news/press-releases/brightedge-data-ai-search-reaching-tipping-point-ai-agents-2026",
    "credibility": "Medium",
    "credibility_rationale": "First-party data from BrightEdge, a large enterprise SEO data vendor, but presented as a press release stating headline percentages without disclosing sample size, site count, measurement window or how 'agent requests' and 'human organic search activity' were defined, so vendor-reported statistics rather than a methodology-documented study.",
    "evidence_type": "statistic",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Anthropic runs three robots.txt-controllable bots (ClaudeBot, Claude-User, Claude-SearchBot) and says blocking Claude-SearchBot 'may reduce your site's visibility' in search.",
    "description": "Anthropic Help Center (April 7, 2026) defines ClaudeBot (training), Claude-User (user-initiated fetches) and Claude-SearchBot (search indexing), says the bots honor robots.txt and Crawl-delay, and states disabling Claude-SearchBot 'may reduce your site's visibility and accuracy in user search results.' IP ranges at claude.com/crawling/bots.json.",
    "date": "2026-04-07",
    "application_area": "Technical Access & Crawling",
    "source": "Anthropic Help Center: Does Anthropic crawl data from the web, and how can site owners block the crawler?",
    "source_link": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
    "credibility": "High",
    "credibility_rationale": "First-party documentation from the operator of the crawlers; quotes verified.",
    "evidence_type": "official-guidance",
    "publisher": "Anthropic",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Google's nosnippet and max-snippet directives also control whether and how much page content can be used as direct input for AI Overviews and AI Mode.",
    "description": "Google Search Central robots meta tag documentation (updated 2026-03-24) extends nosnippet and max-snippet to AI Overviews and AI Mode as the page-level control over use in AI answers without deindexing, except where the publisher grants permission separately; max-snippet:0 equals nosnippet.",
    "date": "2026-03-24",
    "application_area": "Technical Access & Crawling",
    "source": "Google Search Central: Robots meta tag, data-nosnippet, and X-Robots-Tag specifications",
    "source_link": "https://developers.google.com/search/docs/crawling-indexing/robots-meta-tag",
    "credibility": "High",
    "credibility_rationale": "Official Google developer documentation specifying how its own directives affect AI features; carries a 2026-03-24 update stamp.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "The AI bot-to-human visit ratio rose from 1:200 (Q1 2025) to 1:50 (Q2) and 1:31 (Q4 2025); RAG bots scraped 4.5x more than training bots and 30% of scrapes ignored robots.txt.",
    "description": "TollBit's 'State of the Bots' Q3-Q4 2025 report found AI scraping grew at an average quarterly rate of 24.4% from Q2 through year-end, RAG bot activity grew ~33% from Q2 to Q4 2025 while training activity declined 15%, and per page a RAG bot averaged ~10x the scrapes of a training bot. Retrieval-driven demand for content is rising while the referral clicks it returns are shrinking.",
    "date": "2026-02-04",
    "application_area": "Technical Access & Crawling",
    "source": "TollBit: State of the Bots, 2025 Q3 & Q4 - The Leaky Pipes",
    "source_link": "https://tollbit.com/state-of-the-bots/q3-q4-2025/",
    "credibility": "Medium",
    "credibility_rationale": "First-party, methodology-described study by an AI-bot monetization vendor with a large publisher network (6,800+ publishers per TollBit), widely covered by The Register, Digiday and Press Gazette. Rated Medium because TollBit has a commercial stake in the narrative, its network skews to news publishers, and it acknowledges its bot-vs-human detection is conservative.",
    "evidence_type": "study",
    "publisher": "TollBit",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "To be eligible as a supporting link in Google AI Overviews or AI Mode, a page must be indexed and eligible to show in Google Search with a snippet.",
    "description": "Google's documentation: 'To be eligible to be shown as a supporting link in AI Overviews or AI Mode, a page must be indexed and eligible to be shown in Google Search with a snippet.' Preview controls (nosnippet, data-nosnippet, max-snippet, noindex) therefore also restrict AI feature appearance.",
    "date": "2025-12-10",
    "application_area": "Technical Access & Crawling",
    "source": "Google Search Central: AI features and your website",
    "source_link": "https://developers.google.com/search/docs/appearance/ai-features",
    "credibility": "High",
    "credibility_rationale": "Official Google Search Central documentation; sentence confirmed verbatim by spot-check.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "User-action AI crawling grew more than 15x during 2025, GPTBot reached ~7.5% of verified-bot traffic (behind Googlebot at 28%+), and Anthropic's crawl-to-refer ratio hit 500,000:1.",
    "description": "Cloudflare Radar 2025 Year in Review (Jan 1-Dec 2, 2025). GPTBot, ClaudeBot and CCBot received the most full-disallow robots.txt directives among the top 10,000 domains; OpenAI's ratio peaked at 3,700:1 (March) and Perplexity's generally stayed below 400:1 (below 200:1 from September).",
    "date": "2025-12",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: The 2025 Cloudflare Radar Year in Review",
    "source_link": "https://blog.cloudflare.com/radar-2025-year-in-review/",
    "credibility": "High",
    "credibility_rationale": "First-party network-scale annual dataset from a major provider; spot-check confirmed all attributed figures.",
    "evidence_type": "statistic",
    "publisher": "Cloudflare Radar",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Only 10.13% of ~300,000 domains had an llms.txt file, and its presence had no measurable positive effect on AI citation frequency; removing it improved the model.",
    "description": "SE Ranking analyzed ~300,000 domains for llms.txt presence against traffic and domain-level AI citation frequency using Spearman correlation and an XGBoost model with SHAP; llms.txt added noise and predictions improved when it was removed. Authors note results are context-dependent.",
    "date": "2025-11-07",
    "application_area": "Technical Access & Crawling",
    "source": "SE Ranking: Does LLMs.txt impact your AI visibility and citations? No, according to research",
    "source_link": "https://seranking.com/blog/llms-txt/",
    "credibility": "High",
    "credibility_rationale": "Established SEO data vendor; large sample with documented statistical method; consistent with Mueller's statement. Citation sample not fully specified.",
    "evidence_type": "study",
    "publisher": "SE Ranking (Yulia Deda)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Bing supports the data-nosnippet attribute: marked content stays indexed and rankable but is excluded from snippets and AI summaries in Bing Search and Copilot.",
    "description": "Bing Webmaster Blog (Oct 15, 2025; Madhavan, Canel, Milos) gives publishers section-level control over what appears in Bing snippets and Copilot answers; tagged content remains rankable and changes take seconds to a week depending on crawl schedule.",
    "date": "2025-10-15",
    "application_area": "Technical Access & Crawling",
    "source": "Bing Webmaster Blog: Bing Introduces Support for the data-nosnippet HTML Attribute",
    "source_link": "https://blogs.bing.com/webmaster/October-2025/Bing-Introduces-Support-for-the-data-nosnippet-HTML-Attribute",
    "credibility": "High",
    "credibility_rationale": "Official platform announcement bylined by three Microsoft Bing Principal Product Managers.",
    "evidence_type": "official-guidance",
    "publisher": "Microsoft Bing (Bing Webmaster Blog)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Cloudflare's Content Signals Policy adds a robots.txt Content-Signal line (search, ai-input, ai-train), defaulting to 'search=yes, ai-train=no' on 3.8M+ managed domains.",
    "description": "Official Cloudflare announcement: 'ai-input' covers RAG, grounding and real-time use in AI answers; 'search' excludes AI-generated summaries; ai-input is left unset by default. Cloudflare states the signals express preferences and are not technical countermeasures.",
    "date": "2025-09-24",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: Content Signals Policy",
    "source_link": "https://blog.cloudflare.com/content-signals-policy/",
    "credibility": "High",
    "credibility_rationale": "Official policy announcement from Cloudflare about its own robots.txt extension; definitions and defaults confirmed on the page.",
    "evidence_type": "official-guidance",
    "publisher": "Cloudflare (Will Allen)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "In July 2025 Anthropic crawled ~38,066 pages per referral sent, OpenAI 1,091, Perplexity 195 and Google 5.4, with training driving nearly 80% of AI crawler activity.",
    "description": "Cloudflare Radar measurement, Jan-Jul 2025 with YoY comparison. Anthropic's ratio fell from 286,930:1 (Jan) to 38,066:1 (Jul, -86.7%); GPTBot's share of AI crawl traffic rose from 4.7% to 11.7% and ClaudeBot's from 6% to ~10%; training's share rose from 72% to nearly 80%.",
    "date": "2025-08-29",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: The crawl-to-click gap: Cloudflare data on AI bots, training, and referrals",
    "source_link": "https://blog.cloudflare.com/crawlers-click-ai-bots-training/",
    "credibility": "High",
    "credibility_rationale": "First-party Cloudflare Radar network-wide data (Joao Tome, Aug 29, 2025) with a full ratio table; spot-check confirmed all figures.",
    "evidence_type": "statistic",
    "publisher": "Cloudflare (Joao Tome)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Nearly 80% of AI bot crawling in July 2025 was for model training; in the first week of August 2025 crawl-to-refer ratios were 50,000:1 for Anthropic, 887:1 for OpenAI and 118:1 for Perplexity.",
    "description": "Cloudflare classified AI crawler requests by declared purpose and industry for July and early August 2025. User-action and undeclared purposes were under 5% of AI bot traffic, ChatGPT-User was nearly three quarters of user-action requests, and for News & Publications the ratios were 2,500:1 (Anthropic), 152:1 (OpenAI) and 32.7:1 (Perplexity).",
    "date": "2025-08-28",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: A deeper look at AI crawlers: breaking down traffic by purpose and industry",
    "source_link": "https://blog.cloudflare.com/ai-crawler-traffic-by-purpose-and-industry/",
    "credibility": "High",
    "credibility_rationale": "First-party Cloudflare Radar telemetry with classification method described; spot-check confirmed all numbers and corrected the ratio window to the first week of August 2025.",
    "evidence_type": "statistic",
    "publisher": "Cloudflare (David Belson)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Cloudflare documented Perplexity using undeclared crawlers with a generic Chrome user agent and rotating IPs to access robots.txt-blocked content at 3-6 million requests per day.",
    "description": "Cloudflare registered unindexed test domains with restrictive robots.txt and WAF blocks, then queried Perplexity, which returned details of blocked pages. The stealth traffic spanned tens of thousands of domains and Cloudflare de-listed Perplexity as a verified bot. Perplexity publicly denied the findings, attributing traffic to a third-party cloud browser service.",
    "date": "2025-08-04",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: Perplexity is using stealth, undeclared crawlers to evade website no-crawl directives",
    "source_link": "https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/",
    "credibility": "High",
    "credibility_rationale": "First-party investigation with test methodology and volumes documented; widely reported; note it is a contested vendor-versus-vendor dispute.",
    "evidence_type": "analysis",
    "publisher": "Cloudflare (Reid Tatoris)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "AI bot scrapes bypassing robots.txt on publisher sites rose from 3.3% (Q4 2024) to ~13% (Q1 2025), and RAG scrapes per site grew 49% quarter over quarter.",
    "description": "Figures from TollBit's State of the Bots reports (Q4 2024, Q1 2025) drawn from its publisher network, presented in a FIPP webinar (July 8, 2025): 26M+ disallowed scrapes in March 2025, RAG bots growing ~2.5x faster than training bots, traffic from AI answers ~96% lower than an equivalent Google search. TollBit's primary reports are gated.",
    "date": "2025-07-08",
    "application_area": "Technical Access & Crawling",
    "source": "FIPP: State of the Bots: FIPP and TollBit put the spotlight on AI scraping",
    "source_link": "https://www.fipp.com/news/state-of-the-bots-fipp-and-tollbit-put-the-spotlight-on-ai-scraping/",
    "credibility": "Medium",
    "credibility_rationale": "FIPP is the established publishers' trade association and TollBit data is regularly cited by Press Gazette, but this is a webinar write-up with no documented methodology, sample or definitions; primary reports could not be read.",
    "evidence_type": "analysis",
    "publisher": "FIPP (International Federation of Periodical Publishers), reporting TollBit data",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "Anthropic's ClaudeBot made ~70,900 HTML page requests per HTML referral Claude sent back (June 19-26, 2025), while Mistral referred 10x more visits than it crawled.",
    "description": "Cloudflare launched the crawl-to-refer ratio on Radar, comparing HTML crawl requests to HTML referrals carrying each platform's Referer header for June 19-26, 2025. Authors caution Claude's native app sends no Referer header, so the ratio may be overstated; Google's ratio dropped 19.4% WoW on reduced Googlebot activity.",
    "date": "2025-07-01",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: The crawl before the fall... of referrals: understanding AI's impact on content providers",
    "source_link": "https://blog.cloudflare.com/ai-search-crawl-refer-ratio-on-radar/",
    "credibility": "High",
    "credibility_rationale": "First-party Cloudflare Radar network-wide measurement (David Belson, Sam Rhea) with caveats stated; the Radar metric is widely reported.",
    "evidence_type": "statistic",
    "publisher": "Cloudflare",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Only 14% of top-10,000 domains with a robots.txt had AI-bot-specific directives in May 2025, while GPTBot request volume grew 305% year over year to 7.7% of crawler traffic.",
    "description": "Cloudflare analyzed one year of crawler traffic (May 2024-May 2025) and robots.txt files across its top 10,000 domains: 3,816 had a robots.txt, 546 (~14%) included AI-bot directives, 312 disallowed GPTBot (250 fully). Googlebot held 50% of crawler traffic, ClaudeBot fell 46%, ChatGPT-User grew 2,825%.",
    "date": "2025-07-01",
    "application_area": "Technical Access & Crawling",
    "source": "Cloudflare Blog: From Googlebot to GPTBot: Who's crawling your site in 2025",
    "source_link": "https://blog.cloudflare.com/from-googlebot-to-gptbot-whos-crawling-your-site-in-2025/",
    "credibility": "High",
    "credibility_rationale": "First-party Cloudflare network data plus a documented robots.txt analysis; all figures confirmed on the page by the qualifier.",
    "evidence_type": "study",
    "publisher": "Cloudflare (Joao Tome)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Google says more restrictive preview controls (nosnippet, data-nosnippet, max-snippet, noindex) will limit how content is featured in its AI experiences.",
    "description": "John Mueller's Search Central blog post (May 21, 2025): 'Make use of nosnippet, data-nosnippet, max-snippet, or noindex to set your display preferences. More restrictive permissions will limit how your content is featured in our AI experiences.'",
    "date": "2025-05-21",
    "application_area": "Technical Access & Crawling",
    "source": "Google Search Central Blog: Top ways to ensure your content performs well in Google's AI experiences on Search",
    "source_link": "https://developers.google.com/search/blog/2025/05/succeeding-in-ai-search",
    "credibility": "High",
    "credibility_rationale": "Official Google Search Central blog post by John Mueller (Search Relations); reported by Search Engine Land and Search Engine Journal.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central (John Mueller)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Perplexity correctly identified about one-third of excerpts from publishers that blocked its crawler in robots.txt, so crawl directives did not prevent retrieval.",
    "description": "The Tow Center's eight-engine test included publishers that disallow AI crawlers. Despite Perplexity's stated policy of respecting robots.txt, it correctly identified excerpts from blocked publishers such as National Geographic about a third of the time, while some engines that were permitted to crawl still answered wrongly.",
    "date": "2025-03-06",
    "application_area": "Technical Access & Crawling",
    "source": "Columbia Journalism Review / Tow Center for Digital Journalism: AI Search Has a Citation Problem",
    "source_link": "https://www.cjr.org/tow_center/we-compared-eight-ai-search-engines-theyre-all-bad-at-citing-news.php",
    "credibility": "High",
    "credibility_rationale": "Primary research from the Tow Center for Digital Journalism at Columbia University (Jazwinska and Chandrasekar) with a documented method: 20 publishers x 10 articles x 8 chatbots = 1,600 queries, per-tool results reported. Widely cited by major outlets and by the BBC/EBU report.",
    "evidence_type": "study",
    "publisher": "Columbia Journalism Review / Tow Center for Digital Journalism (Columbia University)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "No major non-Google AI crawler (GPTBot, ClaudeBot, PerplexityBot, Bytespider, Meta-ExternalAgent) executes JavaScript, so client-rendered content is invisible to them.",
    "description": "Vercel and MERJ analyzed one month of crawler traffic on Vercel's network (569M GPTBot, 370M Claude, 314M AppleBot, 24.4M PerplexityBot vs 4.5B Googlebot). ChatGPT and Claude fetched JS files (11.50% and 23.84% of requests) but showed no execution; only Gemini (Googlebot infrastructure) and AppleBot render JavaScript.",
    "date": "2024-12-17",
    "application_area": "Technical Access & Crawling",
    "source": "Vercel: The Rise of the AI Crawler",
    "source_link": "https://vercel.com/blog/the-rise-of-the-ai-crawler",
    "credibility": "High",
    "credibility_rationale": "Major infrastructure provider reporting first-party network logs with MERJ; authors include Vercel CTO Malte Ubl; very widely cited. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "Vercel (with MERJ)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "AI crawlers waste crawl budget on dead URLs: 34.82% of ChatGPT fetches and 34.16% of Claude fetches returned 404, versus 8.22% for Googlebot.",
    "description": "Same Vercel/MERJ analysis. ChatGPT prioritized HTML (57.70%) while Claude fetched a large share of images (35.17%), indicating AI crawlers are less efficient at URL selection and benefit from clean sitemaps, redirects and current URLs.",
    "date": "2024-12-17",
    "application_area": "Technical Access & Crawling",
    "source": "Vercel: The Rise of the AI Crawler",
    "source_link": "https://vercel.com/blog/the-rise-of-the-ai-crawler",
    "credibility": "High",
    "credibility_rationale": "Same first-party Vercel/MERJ dataset; spot-check confirmed 404 rates and content shares.",
    "evidence_type": "study",
    "publisher": "Vercel (with MERJ)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "48% of top news sites in 10 countries blocked OpenAI crawlers and 24% blocked Google-Extended by end of 2023 (79% blocking OpenAI in the USA).",
    "description": "Reuters Institute examined Wayback-archived robots.txt for every day of 2023 across the 15 most-used news sources in ten countries. 97% of sites blocking Google-Extended also blocked OpenAI; no site reversed a block during 2023; legacy print publications were most likely to block.",
    "date": "2024-02-22",
    "application_area": "Technical Access & Crawling",
    "source": "Reuters Institute for the Study of Journalism: How Many News Websites Block AI Crawlers? (Richard Fletcher, Factsheet)",
    "source_link": "https://reutersinstitute.politics.ox.ac.uk/sites/default/files/2024-02/Fletcher_How_Many_News_Websites_Block_AI_Crawlers.pdf",
    "credibility": "High",
    "credibility_rationale": "Published factsheet (DOI 10.60625/risj-xm9g-ws87) from the University of Oxford's Reuters Institute by its Director of Research, with fully documented method.",
    "evidence_type": "paper",
    "publisher": "Reuters Institute for the Study of Journalism, University of Oxford",
    "publisher_type": "research-institution"
  },
  {
    "finding": "OpenAI says sites that block OAI-SearchBot will not appear in ChatGPT search answers, robots.txt changes take ~24 hours, and user-initiated ChatGPT-User fetches may ignore robots.txt.",
    "description": "OpenAI crawler documentation lists OAI-SearchBot (opted-out sites 'will not be shown in ChatGPT search answers, though can still appear as navigational links'; ~24 hours to adjust), GPTBot (training), ChatGPT-User ('Because these actions are initiated by a user, robots.txt rules may not apply'; not used for Search eligibility) and OAI-AdsBot, with published IP-range files. Undated page.",
    "date": "unknown",
    "application_area": "Technical Access & Crawling",
    "source": "OpenAI Developers: Overview of OpenAI Crawlers",
    "source_link": "https://developers.openai.com/api/docs/bots",
    "credibility": "High",
    "credibility_rationale": "Official OpenAI developer documentation; spot-check confirmed every quoted statement; no publication date on the page.",
    "evidence_type": "official-guidance",
    "publisher": "OpenAI",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Perplexity says PerplexityBot indexes sites for search results and is not used for model training, while the user-triggered Perplexity-User fetcher 'generally ignores robots.txt rules.'",
    "description": "Perplexity's crawler docs: PerplexityBot surfaces and links websites in Perplexity results and is not used for foundation-model crawling; Perplexity-User supports user actions and 'generally ignores robots.txt rules.' IP ranges published. Cloudflare's Aug 2025 investigation reported undeclared crawlers, so this reflects stated policy, not necessarily observed behavior.",
    "date": "unknown",
    "application_area": "Technical Access & Crawling",
    "source": "Perplexity Docs: Perplexity Crawlers",
    "source_link": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "credibility": "High",
    "credibility_rationale": "Official Perplexity documentation; quotes confirmed; corroborated by Cloudflare's identification of the same two declared agents.",
    "evidence_type": "official-guidance",
    "publisher": "Perplexity",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "AI Overviews on commercial-intent SERPs grew an average of 71% between Nov 2025 and Apr 2026 across 600,000+ US keywords, with Finance up 231.25%.",
    "description": "Semrush analyzed 600,000+ keywords across 10 industries in its US desktop database. Computers & Electronics rose 107.62% and Games 76.66% for commercial queries, while AIOs on transactional-intent queries fell 5% overall (Computers & Electronics +80.45%, News +45.96%); AIOs now appear alongside Google Ads twice as often as a year earlier, after SERPs with both Ads and AIOs grew 394% toward the end of 2025.",
    "date": "2026-07-02",
    "application_area": "Platform Behavior & Guidance",
    "source": "Semrush: AI Overviews are expanding across commercial intent search",
    "source_link": "https://www.semrush.com/blog/ai-overviews-commercial-search-study/",
    "credibility": "High",
    "credibility_rationale": "Semrush is a major SEO data vendor; the post documents its methodology (600,000+ keywords across 10 industries, US desktop database, November 2025 to April 2026), all headline figures were verified on the page, and it comes from the same research team as the widely cited 10M-keyword AIO study.",
    "evidence_type": "study",
    "publisher": "Semrush (Luke Harsel; contributors Anna Yudina, Christine Skopec)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "49% of US adults use AI chatbots (up from 33% in 2024), 42% use them to search for information and 60% read AI-generated summaries in search results.",
    "description": "Pew's Americans and AI 2026 report (5,119 US adults surveyed February 17-23, 2026) found 24% use chatbots daily and 13% use them to get news; ChatGPT leads at 44% followed by Gemini at 24%. Searching for information is the single most common chatbot use.",
    "date": "2026-06-17",
    "application_area": "Platform Behavior & Guidance",
    "source": "Pew Research Center: Americans and AI 2026: Chatbots, Smart Devices and Views on Impact",
    "source_link": "https://www.pewresearch.org/internet/2026/06/17/americans-and-ai-2026-chatbots-smart-devices-and-views-on-impact/",
    "credibility": "High",
    "credibility_rationale": "Pew Research Center is a nonpartisan research institution with a fully documented probability-based survey methodology (American Trends Panel); the survey covered 5,119 US adults fielded February 17-23, 2026, and all reported figures were confirmed on the page.",
    "evidence_type": "statistic",
    "publisher": "Pew Research Center",
    "publisher_type": "research-institution"
  },
  {
    "finding": "GEO-style rewriting raised the rate at which flawed products entered an LLM recommendation agent's shortlist by up to 83.2 points; simple defenses cut harmful promotion by at most 39.2 points.",
    "description": "SafeGEO (Wen, Liu, Liu, Jiao, Yang, Wu, Tang) evaluates 22 GEO attack variants across 600 recommendation cases where the promoted target product is objectively flawed for the user's need. Agent-side defenses reduced harmful target promotion by up to 39.2 percentage points but did not restore no-GEO performance.",
    "date": "2026-06-08",
    "application_area": "Platform Behavior & Guidance",
    "source": "arXiv (Wen, Liu, Liu, Jiao, Yang, Wu, Tang; University of Toronto/ZBot Technology/UC San Diego): SafeGEO: Understanding Generative Engine Optimization Risks in Recommendation Agents",
    "source_link": "https://arxiv.org/abs/2606.28356",
    "credibility": "High",
    "credibility_rationale": "Mixed academic/industry authorship led by University of Toronto researchers with UC San Diego co-authors; a documented evaluation suite (22 GEO attack variants x 600 cases with objectively flawed targets) with quantified attack and defense effects. Very recent preprint (June 2026, revised September 2026), not yet peer-reviewed, so headline numbers are provisional.",
    "evidence_type": "paper",
    "publisher": "arXiv preprint; University of Toronto, ZBot Technology, UC San Diego, Coolwei AI Lab",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google says AI Overviews has 2.5B+ monthly users and AI Mode 1B+, and sites using the new Search Console opt-out 'will not receive traffic or impressions' from AI features.",
    "description": "Mrinalini Loew (GM, Google Search Ecosystem), June 3, 2026, updated Aug 31, 2026: AIOs 2.5B+ MAU, AI Mode 1B+. A new Search Console control lets owners decide whether their site appears in and grounds generative AI Search features; opted-out sites get no AI-feature traffic or impressions, and the control is not a ranking signal elsewhere. Rolled out worldwide Aug 31, 2026.",
    "date": "2026-06-03",
    "application_area": "Platform Behavior & Guidance",
    "source": "Google (The Keyword): New opportunities, control and insights for website owners",
    "source_link": "https://blog.google/products-and-platforms/products/search/new-controls-website-owners/",
    "credibility": "High",
    "credibility_rationale": "Official Google announcement, corroborated by the Search Central blog and SEL/SEJ coverage. Spot-check confirmed figures and dates.",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "AI Overviews appeared on 13.7% of 55,393 trending queries but 64.7% of question-phrased ones; ~30% of cited sources were absent from the SERP and 11% of claims lacked support.",
    "description": "Xu, Iqbal and Montgomery (Washington University in St. Louis) audited AIO activation, source quality, claim fidelity and publisher impact across 55,393 trending queries in 19 categories over 40 days (March 13 to April 21, 2026), with lower activation for politically sensitive topics. AIO-cited domains were more credible on average than first-page organic results, roughly 30% of cited sources did not appear in the accompanying SERP (indicating a selection mechanism distinct from ranking), 11.0% of 98,020 atomic claims lacked support from cited pages with omission the primary failure type, and over half of AIO-cited pages carry display advertising.",
    "date": "2026-05-13",
    "application_area": "Platform Behavior & Guidance",
    "source": "arXiv: Measuring Google AI Overviews: Activation, Source Quality, Claim Fidelity, and Publisher Impact (arXiv 2605.14021)",
    "source_link": "https://arxiv.org/abs/2605.14021",
    "credibility": "High",
    "credibility_rationale": "University research (Washington University in St. Louis; Umar Iqbal is a known web-measurement researcher and Jacob Montgomery a political-science methodologist) with large-scale longitudinal measurement (55,393 queries, 19 categories, 40 days, 98,020 decomposed claims) and documented methodology. Marked 'under review', so not yet peer reviewed.",
    "evidence_type": "paper",
    "publisher": "Washington University in St. Louis (Haofei Xu, Umar Iqbal, Jacob M. Montgomery) via arXiv",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google AI Overviews appeared on 51.5% of 11,500 real-user queries, and their cited sources barely overlap with organic results (Jaccard 0.18) or Gemini (0.11).",
    "description": "Grossman et al. (NJIT, ACM SIGIR 2026) collected source lists from Google SERP, AIO and Gemini 2.5 Flash for 11,500 time-invariant ORCAS queries (14,212 total). Jaccard: AIO vs SERP 0.18, AIO vs Gemini 0.11, SERP vs Gemini 0.16; AIOs were less consistent on repeated queries (RBO 0.69 vs 0.86) and favored Google-owned domains.",
    "date": "2026-04-30",
    "application_area": "Platform Behavior & Guidance",
    "source": "arXiv / ACM SIGIR 2026 (NJIT): How Generative AI Disrupts Search: An Empirical Study of Google Search, Gemini, and AI Overviews",
    "source_link": "https://arxiv.org/html/2604.27790v1",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed (Proceedings of ACM SIGIR 2026); NJIT-led with NTU and Indiana co-authors; hosted on the NJIT faculty site and covered by NJIT News; figures verified in Table 2.",
    "evidence_type": "paper",
    "publisher": "Riley Grossman, Songjiang Liu, Michael K. Chen, Mike Smith, Cristian Borcea, Yi Chen (New Jersey Institute of Technology)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google AI Overviews expanded from 7 to 229 countries (2024-2025), surfacing fewer long-tail sources, less response variety and more low-credibility sources than traditional search.",
    "description": "Aral, Li and Zuo (MIT) ran 24,000 queries in 243 countries generating 2.8M AI and traditional results in 2024 and 2025. Only 1% of Covid queries were AI-answered in 2024 vs 66%+ in 2025; AI search surfaced fewer long-tail sources and more low-credibility and right/center-leaning sources. Preprint.",
    "date": "2026-02-13",
    "application_area": "Platform Behavior & Guidance",
    "source": "arXiv (MIT): The Rise of AI Search: Implications for Information Markets and Human Judgement at Scale",
    "source_link": "https://arxiv.org/abs/2602.13415",
    "credibility": "High",
    "credibility_rationale": "Sinan Aral (MIT Sloan, MIT IDE director) with co-authors; very large documented sample; cited by subsequent arXiv work. Preprint, not yet peer-reviewed.",
    "evidence_type": "paper",
    "publisher": "Sinan Aral, Haiwen Li, Rui Zuo (MIT Sloan / MIT IDE)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "AI Overviews appear on ~48% of BrightEdge-tracked queries (up 58% YoY), average ~1,200px tall, and only ~17% of AIO-cited sources rank in the organic top 10.",
    "description": "BrightEdge Generative Parser (AI Catalyst) data comparing February 2025 to February 2026 across nine verticals. AIO presence peaked above 50%, average height reached ~1,340px in December 2025 against a ~900px desktop viewport, top-100 overlap of cited sources was ~50-53%, and top-10 overlap by industry was Healthcare 24.0%, Education 23.1%, B2B Tech 22.6%, Entertainment 18.5% (from 3.2%), Travel 17.7% (from 5.7%), eCommerce 13.4% (from 2.9%), Finance 11.3% and Restaurants 9.3%.",
    "date": "2026-02-12",
    "application_area": "Platform Behavior & Guidance",
    "source": "BrightEdge: AI Overviews at the One-Year Mark: Presence, Size, and What They're Citing",
    "source_link": "https://www.brightedge.com/resources/weekly-ai-search-insights/ai-overviews-one-year-presence-size-citing",
    "credibility": "High",
    "credibility_rationale": "BrightEdge is a large enterprise SEO data vendor whose Generative Parser / AI Catalyst tracks AIOs continuously across nine verticals; the article documents its data source, comparison window and per-industry figures, and BrightEdge AIO data is routinely cited by Search Engine Land and Search Engine Journal. No named author on the page is a minor transparency gap.",
    "evidence_type": "analysis",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "OpenAI's ChatGPT shopping feed spec requires nine fields (item_id, title, description, url, brand, seller_name, image_url, availability, price); onboarding is limited to approved partners.",
    "description": "OpenAI's Agentic Commerce developer docs accept merchant-submitted JSONL, CSV and TSV feeds (plus gzip variants), recommend a full daily snapshot via file upload with API upserts throughout the day, and state that is_eligible_search defaults to true (eligibility does not guarantee display) while is_eligible_checkout opts in only when search eligibility is also true. OpenAI may remove a product or ban a seller from being surfaced in ChatGPT for policy violations.",
    "date": "2026",
    "application_area": "Platform Behavior & Guidance",
    "source": "OpenAI Developers: Agentic Commerce Product Feed Specification",
    "source_link": "https://developers.openai.com/commerce/specs/feed",
    "credibility": "High",
    "credibility_rationale": "First-party official developer documentation from OpenAI, the platform operating ChatGPT shopping and the Agentic Commerce Protocol; the spec page and companion Get Started guide contain the quoted statements verbatim.",
    "evidence_type": "official-guidance",
    "publisher": "OpenAI (Developers documentation)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Across 10M+ keywords, the share of US queries triggering AI Overviews rose from 6.49% (Jan 2025) to a 24.61% peak (Jul 2025) before settling at 15.69% (Nov 2025).",
    "description": "Semrush tracked 10M+ keywords monthly from January through November 2025. From October 2024 to October 2025 the intent mix shifted: commercial queries triggering AIOs rose from 8.15% to 18.57%, transactional from 1.98% to 13.94% and navigational from 0.84% to 10.33%, while the informational share of AIO-triggering queries fell from 91.3% to 57.1%; on 200K+ identical keywords before and after they began showing AIOs, the zero-click rate fell from 33.75% to 31.53%.",
    "date": "2025-12-15",
    "application_area": "Platform Behavior & Guidance",
    "source": "Semrush: AI Overviews Study (10M keywords)",
    "source_link": "https://www.semrush.com/blog/semrush-ai-overviews-study",
    "credibility": "High",
    "credibility_rationale": "Semrush is a major SEO data vendor; the study documents its method (10M+ keywords tracked monthly through 2025; 200K+ identical keywords for the zero-click before/after comparison), all figures were verified on the page, and Semrush AIO studies are widely cited across the industry.",
    "evidence_type": "study",
    "publisher": "Semrush (Jana Garanko; contributors Luke Harsel, Anna Yudina, Aleksandr Drozdov, Christine Skopec)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Google states there are no additional requirements or special optimizations needed to appear in AI Overviews or AI Mode; standard SEO best practices apply.",
    "description": "Google Search Central 'AI features and your website' (last updated 2025-12-10) states verbatim: 'There are no additional requirements to appear in AI Overviews or AI Mode, nor other special optimizations necessary,' adding that SEO best practices remain relevant because these features are rooted in core ranking systems.",
    "date": "2025-12-10",
    "application_area": "Platform Behavior & Guidance",
    "source": "Google Search Central: AI features and your website",
    "source_link": "https://developers.google.com/search/docs/appearance/ai-features",
    "credibility": "High",
    "credibility_rationale": "Official documentation from the platform operating AI Overviews and AI Mode; spot-check confirmed the quoted sentence and the 2025-12-10 update stamp.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Google says robots.txt directives for Googlebot are the control for AI features in Search, and AI-feature traffic is counted in Search Console's 'Web' search type.",
    "description": "Same Google documentation: robots.txt directives for Googlebot are the control for how sites are crawled for Search; sites appearing in AI features are included in overall search traffic in Search Console (Web search type); Google-Extended limits AI training and grounding in some other Google systems; AIOs and AI Mode may use query fan-out.",
    "date": "2025-12-10",
    "application_area": "Platform Behavior & Guidance",
    "source": "Google Search Central: AI features and your website",
    "source_link": "https://developers.google.com/search/docs/appearance/ai-features",
    "credibility": "High",
    "credibility_rationale": "Official Google Search Central documentation; statements confirmed by spot-check.",
    "evidence_type": "official-guidance",
    "publisher": "Google Search Central",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Only 9% of US adults get news from AI chatbots often or sometimes and 75% never do; half of chatbot news users at least sometimes see news they think is inaccurate.",
    "description": "Pew Research Center surveyed 5,153 US adults on August 18-24, 2025. Fewer than 1% prefer to get news from chatbots over other sources; one-third of chatbot news users find it difficult to determine what is true and 42% are unsure, with younger users reporting inaccuracy more often (59% of 18-29s vs 36% of 65+).",
    "date": "2025-10-01",
    "application_area": "Platform Behavior & Guidance",
    "source": "Pew Research Center: Relatively few Americans are getting news from AI chatbots like ChatGPT",
    "source_link": "https://www.pewresearch.org/short-reads/2025/10/01/relatively-few-americans-are-getting-news-from-ai-chatbots-like-chatgpt/",
    "credibility": "High",
    "credibility_rationale": "Pew Research Center is a nonpartisan research organization using its probability-based American Trends Panel; this analysis surveyed 5,153 US adults with published methodology and topline, and Pew's news-consumption data is routinely cited by major outlets and academics.",
    "evidence_type": "statistic",
    "publisher": "Pew Research Center",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google's John Mueller: 'FWIW no AI system currently uses llms.txt'; consumer LLMs fetch pages for training and grounding but none fetch the llms.txt file.",
    "description": "Posted by John Mueller on Bluesky on June 17, 2025 and reported by Search Engine Roundtable on June 18, 2025; the qualifier verified the primary post text via the public Bluesky API.",
    "date": "2025-06-18",
    "application_area": "Platform Behavior & Guidance",
    "source": "Search Engine Roundtable: Google: No AI System Currently Uses LLMs.txt",
    "source_link": "https://www.seroundtable.com/google-ai-llms-txt-39607.html",
    "credibility": "High",
    "credibility_rationale": "Underlying statement is from Google Search Advocate John Mueller (platform authority) and the primary Bluesky post was independently verified; Search Engine Roundtable (Barry Schwartz) is a long-established search-news outlet that reproduces the quote accurately.",
    "evidence_type": "official-guidance",
    "publisher": "Search Engine Roundtable (Barry Schwartz), reporting John Mueller (Google)",
    "publisher_type": "news-or-media"
  },
  {
    "finding": "In 2025, 7% of people across 48 markets used AI chatbots for news weekly (15% of under-25s), and only 4% had accessed ChatGPT for news in the last week.",
    "description": "The Reuters Institute Digital News Report 2025 established the baseline for chatbot news consumption. Audiences expected AI to make news cheaper to produce (+29 net) and more up-to-date (+16) but less transparent (-8), less accurate (-8) and less trustworthy (-18), with more scepticism in Europe than the US and less in Asia (India 18% weekly chatbot news use vs UK 3%).",
    "date": "2025-06-17",
    "application_area": "Platform Behavior & Guidance",
    "source": "Reuters Institute for the Study of Journalism: Digital News Report 2025 - Overview and key findings (executive summary)",
    "source_link": "https://reutersinstitute.politics.ox.ac.uk/digital-news-report/2025/dnr-executive-summary",
    "credibility": "High",
    "credibility_rationale": "University of Oxford research institute; the 2025 Digital News Report is based on YouGov surveys across 48 markets on six continents with a published methodology and is the standard reference for cross-national news consumption data.",
    "evidence_type": "study",
    "publisher": "Reuters Institute for the Study of Journalism, University of Oxford",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google's AI Mode uses a 'query fan-out' technique that breaks a question into subtopics and issues many searches simultaneously; Deep Search can issue hundreds.",
    "description": "Elizabeth Reid (VP, Head of Search) at Google I/O, May 20, 2025: AI Mode 'uses our query fan-out technique, breaking down your question into subtopics and issuing a multitude of queries simultaneously'; Deep Search 'can issue hundreds of searches.' Search Central documentation confirms both AIO and AI Mode may use fan-out.",
    "date": "2025-05-20",
    "application_area": "Platform Behavior & Guidance",
    "source": "Google (The Keyword): AI in Search: Going beyond information to intelligence (AI Mode updates from Google I/O 2025)",
    "source_link": "https://blog.google/products-and-platforms/products/search/google-search-ai-mode-update/",
    "credibility": "High",
    "credibility_rationale": "First-party announcement by Google's VP of Search, restated in Search Central documentation and widely reported.",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "Google reported AI Overviews reached over 1.5 billion users in 200 countries and drove over 10% query growth for the query types that show them in the U.S. and India.",
    "description": "Sundar Pichai's Google I/O 2025 keynote (May 20, 2025): AIOs 'have scaled to over 1.5 billion users' in 200 countries and territories, and in the U.S. and India 'are driving over 10% growth in the types of queries that show them.' AI Mode rolled out to all U.S. users that day.",
    "date": "2025-05-20",
    "application_area": "Platform Behavior & Guidance",
    "source": "Google (The Keyword): Google I/O 2025: Sundar Pichai's opening keynote",
    "source_link": "https://blog.google/innovation-and-ai/technology/ai/io-2025-keynote/",
    "credibility": "High",
    "credibility_rationale": "Official keynote transcript by Google's CEO on Google's blog; first-party usage figures widely reported.",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "An AI Overview occupies 41.9% of above-the-fold screen height on desktop and 47.9% on mobile; with a co-occurring featured snippet (60.5% of cases) that rises to 67.1-75.7%.",
    "description": "Botify and DemandSphere pixel-depth analysis across 120,778 US keywords (August-September 2024) measured a collapsed AIO at 394.44px desktop / 400px mobile and 632.19px / 632px with a featured snippet. The report says positions 1-3 may earn the CTR of positions 4-6 when an AIO is present, citing internal Botify e-commerce data that CTR drops ~50% between position 1 and 3, and DemandSphere's companion page states organic CTR drops from 15% to 8% when an AIO appears.",
    "date": "2024",
    "application_area": "Platform Behavior & Guidance",
    "source": "Botify x DemandSphere: AI Overviews Study - Inside Google's New Search Reality",
    "source_link": "https://lp.botify.com/hubfs/White%20paper/2024/Botify%20x%20DemandSphere%20-%20AI%20Overviews%20Report.pdf",
    "credibility": "Medium",
    "credibility_rationale": "Joint white paper from two established enterprise SEO platforms with a large documented sample (120,778 US keywords, 57,263 AIO SERPs, 22 websites) and described pixel-depth method. Rated Medium because the keyword set was deliberately curated to encourage AIO appearance, pixel measurements were desktop-only, and the CTR-impact claims rest on internal, undocumented Botify data and a modeled inference.",
    "evidence_type": "study",
    "publisher": "Botify and DemandSphere",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit, YouTube and LinkedIn are the three most-cited domains across 30M AI search sources, with Reddit and YouTube in the top five on all five engines studied.",
    "description": "Peec AI counted sources cited by ChatGPT, Google AI Mode, Gemini, Perplexity and AI Overviews in the US (30M sources; timeframe not stated). Google surfaces favor social content, ChatGPT editorial sources (Wikipedia, Forbes), Perplexity includes G2; patterns vary by industry. Ranked lists only, no percentages.",
    "date": "2026-09-18",
    "application_area": "Third-party & UGC Platforms",
    "source": "Peec AI: Top domains cited by AI search: Analysis based on 30M sources",
    "source_link": "https://peec.ai/blog/top-domains-cited-by-ai-search-analysis-based-on-30m-sources",
    "credibility": "Medium",
    "credibility_rationale": "Venture-backed AI-search analytics vendor (Series A, 3,000+ customers) picked up by Search Engine Land and Contently, but the page gives no percentages, timeframe or prompt-set description.",
    "evidence_type": "study",
    "publisher": "Peec AI",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit (17.9%) and YouTube (17.8%) together take ~35.7% of Google AI Mode citation share, with Facebook (10.2%), Instagram (5.7%) and Quora (3.0%) in the top 10 and Wikipedia at 4.0%.",
    "description": "Ahrefs (Si Quan Ong) analyzed every domain Google AI Mode cited across a broad Brand Radar sample of real US queries. Top 10: reddit.com 17.9%, youtube.com 17.8%, google.com 12.5%, facebook.com 10.2%, instagram.com 5.7%, en.wikipedia.org 4.0%, quora.com 3.0%, amazon.com 3.0%, tiktok.com 2.7%, walmart.com 1.5% (mention share = share of summed top-source citations).",
    "date": "2026-09-02",
    "application_area": "Third-party & UGC Platforms",
    "source": "Ahrefs: The Most Cited Domains in Google AI Mode",
    "source_link": "https://ahrefs.com/blog/most-cited-domains-ai-mode",
    "credibility": "High",
    "credibility_rationale": "Major SEO data vendor using its Brand Radar dataset with the metric defined; already cited by third parties per Ahrefs backlink data.",
    "evidence_type": "study",
    "publisher": "Ahrefs (Si Quan Ong)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "After AI Overviews expanded (Aug 2024), daily comments in Reddit SFW communities rose 12.0% and commenting users 12.4% vs NSFW, gains later erased by AI Mode (Aug 2025).",
    "description": "Zhang, Cui and Zhang (Emory / WashU) used a difference-in-differences design exploiting that SFW but not NSFW subreddits can be surfaced in AIOs, on 105,012 subreddits tracked daily Jan 2024-Jun 2025. Gains concentrated in experience-based discussions and were sharply attenuated after AI Mode launched. Working paper.",
    "date": "2026-05-14",
    "application_area": "Third-party & UGC Platforms",
    "source": "arXiv (Emory University / Washington University in St. Louis): The Impact of AI Search on the Online Content Ecosystem: Evidence from Google and Reddit (Zhang, Cui, Zhang)",
    "source_link": "https://arxiv.org/abs/2605.16428",
    "credibility": "High",
    "credibility_rationale": "Established business-school scholars (Ruomeng Cui, Dennis Zhang); large natural-experiment design (~57.4M subreddit-day observations, fixed effects). Working paper, not yet peer-reviewed.",
    "evidence_type": "paper",
    "publisher": "Peibo Zhang, Ruomeng Cui (Emory University); Dennis J. Zhang (Washington University in St. Louis)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Among 89,000 AI-cited LinkedIn URLs, 500-2,000-word articles and 50-299-word feed posts captured the largest citation shares, and 54-64% of cited posts shared practical advice.",
    "description": "Semrush analyzed 89,000 LinkedIn URLs cited by ChatGPT Search, Google AI Mode and Perplexity across 325,000 prompts in 12 industries (Jan-Feb 2026). Articles were 50-66% of LinkedIn citations, feed posts 15-28%, reshares 5%; authors under 500 followers were cited at equal or higher rates.",
    "date": "2026-03-10",
    "application_area": "Third-party & UGC Platforms",
    "source": "Semrush: We Analyzed 89K LinkedIn URLs Cited in AI Search: Here's What Drives Visibility",
    "source_link": "https://www.semrush.com/blog/linkedin-ai-visibility-study/",
    "credibility": "High",
    "credibility_rationale": "Large data vendor with documented method; corroborated by Inc., PPC Land and others. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "Semrush",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "LinkedIn is referenced in 14.3% of ChatGPT Search responses, 13.5% of Google AI Mode responses and 5.3% of Perplexity responses (11% average).",
    "description": "Same Semrush LinkedIn study (325,000 prompts, Jan-Feb 2026). ChatGPT Search and AI Mode drew 59% of LinkedIn citations from individual creators while Perplexity drew 59% from Company Pages; median cited posts have 15-25 reactions and at most 1 comment.",
    "date": "2026-03-10",
    "application_area": "Third-party & UGC Platforms",
    "source": "Semrush: We Analyzed 89K LinkedIn URLs Cited in AI Search: Here's What Drives Visibility",
    "source_link": "https://www.semrush.com/blog/linkedin-ai-visibility-study/",
    "credibility": "High",
    "credibility_rationale": "Same documented Semrush study with third-party corroboration; spot-check confirmed the 14.3%/13.5%/5.3% figures.",
    "evidence_type": "study",
    "publisher": "Semrush",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "YouTube's citation share is 38.7% on Perplexity, 36.6% on Google AI Overviews, 19.6% on AI Mode and 4.4% on ChatGPT, and 94% of cited YouTube content is long-form video.",
    "description": "OtterlyAI (Rick Tousseyn) examined 100M+ AI citation instances over 30 days across ChatGPT, Google AIO, AI Mode, Perplexity, Copilot (0.5%) and Gemini (0.2%). Long-form 94%, Shorts 5.7%; YouTube was 31.8% of social-media citations, second to Reddit at 46.4%.",
    "date": "2026-03-02",
    "application_area": "Third-party & UGC Platforms",
    "source": "OtterlyAI: YouTube AI Citation Study 2026",
    "source_link": "https://otterly.ai/blog/youtube-ai-citation-study-2026/",
    "credibility": "Medium",
    "credibility_rationale": "Smaller monitoring vendor but large sample with method described and coverage by Inc., Search Engine Land and others; prompt set not fully disclosed.",
    "evidence_type": "study",
    "publisher": "Otterly.AI (Rick Tousseyn)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit accounts for 2.4% of all ChatGPT citations (YouTube 0.99%, LinkedIn 0.39%), and 99% of Reddit citations point to individual discussion threads.",
    "description": "Profound parsed ~700,000 social-platform citations from ChatGPT responses to U.S. English users (Oct-Dec 2025) into URL archetypes. Facebook 0.39%, Instagram 0.23%; ~85% of YouTube citations were specific videos; LinkedIn split across profiles (25.4%), posts (25.1%), company pages (18.4%) and Pulse articles (12.5%).",
    "date": "2026-02-09",
    "application_area": "Third-party & UGC Platforms",
    "source": "Profound: How ChatGPT Cites Social Media",
    "source_link": "https://www.tryprofound.com/blog/chatgpt-reddit-youtube-citations",
    "credibility": "High",
    "credibility_rationale": "Major data vendor with sample, period and method described and named technical-staff authors; figures confirmed on the page.",
    "evidence_type": "study",
    "publisher": "Profound (Brandon Punturo, Sartaj Rajpal)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Community platforms captured 52.5% of AI citations vs 47.5% for brand domains; Google AI Overviews favored brand domains most (59.8%) vs ChatGPT 44.7% and Perplexity 28.9%.",
    "description": "Otterly.AI analyzed 1M+ citations across ChatGPT, Perplexity and Google AIO in Jan-Feb 2026 (Thomas Peham). Perplexity leaned hardest on Reddit/forums (16.9%); the report also claims 73% of sites have technical barriers to AI crawlers.",
    "date": "2026-02-01",
    "application_area": "Third-party & UGC Platforms",
    "source": "Otterly.AI: The AI Citation Economy: What 1+ Million Data Points Reveal About Visibility in 2026",
    "source_link": "https://otterly.ai/blog/the-ai-citations-report-2026/",
    "credibility": "Medium",
    "credibility_rationale": "Smaller monitoring vendor; 1M+ sample stated but prompt set and sampling undocumented; no major trade-press coverage found; page framing of the 52.5% split is internally inconsistent (spot-check).",
    "evidence_type": "study",
    "publisher": "Otterly.AI",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit's share of AI citations grew at least 73% in every commercial category from October 2025 to January 2026; on Perplexity, Reddit alone was 24% of January citations.",
    "description": "Tinuiti's Q1 2026 AI Citation Trends Report (Jen Cornwell) used Profound to track commercial prompts across seven engines and nine product categories. On Perplexity 31% of January citations were social (Reddit 24%, YouTube 3%); Reddit share was above 5% on ChatGPT vs 0.1% on Gemini. Prompt count not disclosed.",
    "date": "2026-02",
    "application_area": "Third-party & UGC Platforms",
    "source": "Tinuiti: Q1 2026 AI Citation Trends Report",
    "source_link": "https://tinuiti.com/research-insights/research/ai-citation-trends-report-q1-2026/",
    "credibility": "Medium",
    "credibility_rationale": "Large performance agency using Profound data with engines, categories and dates documented, but prompt count and publication date absent and no major press coverage.",
    "evidence_type": "study",
    "publisher": "Tinuiti (Jen Cornwell), data via Profound",
    "publisher_type": "consultancy-or-agency"
  },
  {
    "finding": "Review platforms are cited in 34.5% of commercial software AI Overviews yet lost 76.5-92.2% of estimated organic traffic from early 2024 to end of 2025.",
    "description": "SE Ranking analyzed 30,000 commercial software keywords (22,729 with AIOs, captured Dec 1, 2025) across 23 review platforms. Five platforms took 88% of review-platform links; estimated organic traffic fell 84.5% (G2), 89% (Capterra), 92.2% (TrustRadius), 76.5% (Gartner Peer Insights). 'Review' queries cited a review platform 49% of the time vs 17.1% for 'best/top'.",
    "date": "2026-01-29",
    "application_area": "Third-party & UGC Platforms",
    "source": "SE Ranking: Despite 90% Traffic Loss, Review Platforms Top AI Overview Citations",
    "source_link": "https://seranking.com/blog/review-platforms-in-ai-overviews/",
    "credibility": "High",
    "credibility_rationale": "Established SEO platform with recurring, methodology-documented AIO studies regularly reported by Search Engine Land; figures echoed by third parties. Traffic-loss figures are SE Ranking estimates. Spot-check confirmed all figures.",
    "evidence_type": "study",
    "publisher": "SE Ranking",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit is the most-cited domain across answer engines at 3.11% of 4B+ tracked AI citations, ranking #1 on Perplexity and #2 on ChatGPT, AI Overviews and Grok, but #31 on Copilot.",
    "description": "Profound analyzed 4B+ citations and 300M responses (Aug 2024-late Oct 2025) across ChatGPT, Google AIO, AI Mode, Perplexity, Grok and Copilot. Reddit ranked #3 on AI Mode; the average cited Reddit post was one year old, 4% dated 2019 or earlier; purchase-intent subreddits were heavily cited.",
    "date": "2025-11-10",
    "application_area": "Third-party & UGC Platforms",
    "source": "Profound: The Data on Reddit and AI Search",
    "source_link": "https://www.tryprofound.com/blog/the-data-on-reddit-and-ai-search",
    "credibility": "High",
    "credibility_rationale": "Major answer-engine data vendor; very large documented sample with per-engine breakdowns and named authors; figures confirmed on the page by the qualifier.",
    "evidence_type": "study",
    "publisher": "Profound (Josh Blyskal, Sartaj Rajpal)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "ChatGPT's citation of Reddit fell from ~60% of responses in early August 2025 to ~10% by mid-September, while Google AI Mode cited LinkedIn in nearly 15% of responses.",
    "description": "Semrush analyzed 230,000+ prompts and 100M+ citations across ChatGPT search, Google AI Mode and Perplexity (July 14-Oct 12, 2025) with weekly top-25 domain snapshots. Wikipedia in ChatGPT fell from ~55% to under 20%; AI Mode cited Wikipedia in ~2% and favored Google-owned or partner properties. Semrush attributes the drop mainly to ChatGPT reducing over-citation.",
    "date": "2025-11-10",
    "application_area": "Third-party & UGC Platforms",
    "source": "Semrush: The Most-Cited Domains in AI: A 3-Month Study",
    "source_link": "https://www.semrush.com/blog/most-cited-domains-ai/",
    "credibility": "High",
    "credibility_rationale": "Major data vendor; large documented sample with weekly snapshots and named authors; figures confirmed on the page.",
    "evidence_type": "study",
    "publisher": "Semrush (Luke Harsel; contributors Aleksandr Drozdov, Christine Skopec)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Reddit links appear in 12.6% of ChatGPT search answers, 9% of Google AI Mode and 3.5% of Perplexity answers, and 80% of cited Reddit posts have fewer than 20 upvotes.",
    "description": "Semrush analyzed 217,000 prompts across three AI tools and 248,000 cited Reddit URLs. 70% of cited posts had under 20 comments, the median post was ~900 days old and ~80 words, Q&A threads were over 50% of citations; average citation position #6.7 (ChatGPT), #8.8 (AI Mode), #3.4 (Perplexity).",
    "date": "2025-11-10",
    "application_area": "Third-party & UGC Platforms",
    "source": "Semrush: Reddit AI Search Visibility Study",
    "source_link": "https://www.semrush.com/blog/reddit-ai-search-visibility-study/",
    "credibility": "High",
    "credibility_rationale": "Major data vendor; large documented sample with engagement, age, length and format breakdowns; figures confirmed on the page.",
    "evidence_type": "study",
    "publisher": "Semrush (Margarita Loktionova; contributors Cecilia Meis, Aleksandr Drozdov)",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "UGC and community platforms drive nearly half of all AI search citations; Reddit appears in ~22% of answers, and 88% of Reddit citations come from non-brand category queries.",
    "description": "AirOps analyzed 5.5M+ LLM responses across 61K+ queries on ChatGPT, Perplexity, Gemini and Google AI Mode over 23 days (821K+ cited domains). Reddit ranked #1 in Perplexity and AI Mode and #2 in ChatGPT; 75% of YouTube citations came from non-branded queries; LinkedIn ranked #2 in AI Mode.",
    "date": "2025-11-06",
    "application_area": "Third-party & UGC Platforms",
    "source": "AirOps: The Community Flywheel: How Reddit, YouTube, and LinkedIn Decide Who Wins in AI Search",
    "source_link": "https://www.airops.com/report/the-impact-of-ugc-and-community-in-ai-search",
    "credibility": "Medium",
    "credibility_rationale": "GEO software vendor with a commercial interest, but a large documented sample; cited by Search Engine Land and others. Query-selection method not fully transparent.",
    "evidence_type": "study",
    "publisher": "AirOps",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "G2's AI-answer visibility rose from 6.3% (Aug 2025) to 14.9% (Oct 2025); Profound says ~75% of Perplexity's and ~33% of ChatGPT's and AI Overviews' review-site citations come from G2.",
    "description": "G2's Tech Signals report (Nov 5, 2025, Kamaljeet Kalsi) combines three months of first-party visibility tracking with third-party data from Profound, Semrush, Radix and PromptWatch; the Profound figures are a quote from Trevor Pyle (Profound), a G2 commercial partner, and no underlying dataset was located.",
    "date": "2025-11-05",
    "application_area": "Third-party & UGC Platforms",
    "source": "G2 Learn: Tech Signals - Does G2 Get Ranked in AI LLM Search?",
    "source_link": "https://learn.g2.com/tech-signals-does-g2-get-ranked-in-ai-llm-search",
    "credibility": "Low",
    "credibility_rationale": "Promotional first-party reporting on G2's own visibility; the Profound 75%/33% figures are an unpublished quote from a commercial partner with no dataset, and the other third-party data is secondhand. Spot-check confirmed the page and quote.",
    "evidence_type": "analysis",
    "publisher": "G2 (Kamaljeet Kalsi)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "YouTube is cited 200x more than any other video platform by AI engines, averaging 20% citation share: 29.5% in Google AI Overviews, 16.6% AI Mode, 9.7% Perplexity, 0.2% ChatGPT.",
    "description": "BrightEdge monitored YouTube citations across Google AIO, AI Mode, ChatGPT and Perplexity from May 2024 to Sept 2025 via AI Catalyst. Vimeo and TikTok registered 0.1% only in AIOs; other video platforms registered zero. Number of queries sampled is not stated.",
    "date": "2025-09-25",
    "application_area": "Third-party & UGC Platforms",
    "source": "BrightEdge: AI Engines Choose YouTube 200x More Than Any Other Video Platform",
    "source_link": "https://www.brightedge.com/resources/weekly-ai-search-insights/youtube-presence-ai-search",
    "credibility": "High",
    "credibility_rationale": "Large enterprise SEO vendor with monitoring window and platform documented; cited by Forbes, Yahoo/Stacker, AdExchanger and others. Sample size not stated. Spot-check confirmed figures.",
    "evidence_type": "study",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Search Console's Generative AI performance report shows AI Overviews and AI Mode impressions by page, country, device and date, but reports no clicks or queries.",
    "description": "Search Console Help (announced June 3, 2026; worldwide Aug 31, 2026): impressions are how many times links to a site were shown in a generative AI feature, grouped by pages, countries, dates and devices using the canonical URL; no click or query dimensions; data is included in the Web search type of the Performance report.",
    "date": "2026-06",
    "application_area": "Measurement & Methodology",
    "source": "Google Search Console Help: Generative AI performance report (Search)",
    "source_link": "https://support.google.com/webmasters/answer/16984139?hl=en",
    "credibility": "High",
    "credibility_rationale": "Official Google product documentation, corroborated by the Search Central announcement and SEJ/SEL coverage.",
    "evidence_type": "official-guidance",
    "publisher": "Google",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "GEO-Bench found black-box LLM content rewriting matches or beats white-box gradient attacks (STS, RAF, StealthRank) at rank promotion while producing more fluent text.",
    "description": "Nimase, Chen, Qi, Zhao and Hu benchmarked ranking-manipulation attacks under unified protocols with Llama-3.1-8B-Instruct as the ranker across five datasets (Ragroll, STSData, RewriteToRank, LLM Rank Optimizer, C-SEO Bench), comparing black-box prompt attacks (TAP, zero-shot rewriting), white-box gradient attacks and ten 'white-hat' C-SEO strategies on effectiveness (NRG, Success@alpha, Promote@alpha) and stealth (keyword violation rate, perplexity ratio). The three highest average-NRG methods (TAP, Authoritative, Content Improvement) were all black-box, gradient methods ranked in the middle and bottom, and effectiveness traded off against stealth across adversarial attacks.",
    "date": "2026-05-27",
    "application_area": "Measurement & Methodology",
    "source": "arXiv (Nimase, Chen, Qi, Zhao, Hu; USC/ASU): GEO-Bench: Benchmarking Ranking Manipulation in Generative Engine Optimization",
    "source_link": "https://arxiv.org/abs/2605.29107",
    "credibility": "High",
    "credibility_rationale": "University research (USC and ASU; Yue Zhao's lab authored the earlier StealthRank paper) with fully specified method: unified protocols on Llama-3.1-8B-Instruct, five named datasets, explicit effectiveness and stealth metrics. Very recent preprint (May 2026), not yet peer-reviewed or widely cited.",
    "evidence_type": "paper",
    "publisher": "arXiv preprint; University of Southern California and Arizona State University",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google Merchant Center's AI performance insights report share of voice for conversational shopping queries on AI Mode and AI Overviews, for English queries in US, CA, AU, IN and NZ.",
    "description": "Google's Merchant Center help pages (announcement May 27, 2026) describe a report covering organic AI traffic only (not paid ads) that shows share of voice versus competitors, frequency of search types, terms, intents and attributes, products showing for top terms, shopping stages (Discovery, Evaluation, Ready to Buy) and popular attributes missing structured specifications such as size, color or material. Historical data is updated daily with a few days' lag, and Google frames it as a way to optimize product data for AI-powered experiences.",
    "date": "2026-05-27",
    "application_area": "Measurement & Methodology",
    "source": "Google Merchant Center Help: About AI performance insights",
    "source_link": "https://support.google.com/merchants/answer/17200695",
    "credibility": "High",
    "credibility_rationale": "First-party official documentation from Google, the platform operator, describing its own Merchant Center reporting feature; the companion announcement page is dated May 27, 2026 and was independently covered by Search Engine Land, Search Engine Roundtable, PPC Land and Semrush with matching descriptions.",
    "evidence_type": "official-guidance",
    "publisher": "Google (Merchant Center Help)",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "A SIGIR 2026 reproducibility study found 'lost in the middle' position effects in RAG vary significantly across models and datasets, with topic sampling able to mask or exaggerate them.",
    "description": "Gabín, Pérez and Parapar revisited claims that document position and context size drive RAG answer quality under a controlled evaluation framework with contemporary LLMs. Findings on position effects were inconsistent across models, datasets and protocols, topic set size affected whether ordering effects appeared, both factors interacted strongly with retrieval quality and model choice, and discrepancies with an industry study traced to limited topic coverage and reliance on LLM judges; the authors release a calibration procedure, code and configurations.",
    "date": "2026-05-26",
    "application_area": "Measurement & Methodology",
    "source": "arXiv / ACM SIGIR 2026: Lost in the Evidence? Reproducing Document Position and Context Size Effects in RAG (2605.27105)",
    "source_link": "https://arxiv.org/abs/2605.27105",
    "credibility": "High",
    "credibility_rationale": "Peer-reviewed reproducibility paper accepted at ACM SIGIR 2026 (ACM DOI 10.1145/3805712.3808569, listed in the SIGIR 2026 proceedings), authored by the CITIC information-retrieval group at Universidade da Coruña, with code and configurations released.",
    "evidence_type": "paper",
    "publisher": "arXiv; Proceedings of the 49th International ACM SIGIR Conference (Jorge Gabín, Anxo Pérez, Javier Parapar; CITIC, Universidade da Coruña)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Identical AI-search prompts run within 24 hours share only 33-48% of brands (Jaccard 0.33-0.48), so reliable visibility estimates need 7+ repeated runs over 2-4 weeks.",
    "description": "Schulte, Bleeker and Kaufmann (University of St. Gallen) ran eight prompts per campaign across four industries on multiple engines with repeated runs. 24-hour brand Jaccard 0.477 (Consumer Electronics) to 0.327 (Sporting Goods); ChatGPT lowest source overlap (0.233) vs Gemini (0.505). Per-brand SE fell below 0.10 at n=7 runs; a 2-4 week window is recommended.",
    "date": "2026-04-08",
    "application_area": "Measurement & Methodology",
    "source": "arXiv (University of St. Gallen): Don't Measure Once: Measuring Visibility in AI Search (GEO)",
    "source_link": "https://arxiv.org/html/2604.07585",
    "credibility": "High",
    "credibility_rationale": "University of St. Gallen researchers with fully documented design and SE convergence tables; not yet peer-reviewed but the specific numbers are reproduced by multiple industry write-ups.",
    "evidence_type": "paper",
    "publisher": "Julius Schulte, Malte Bleeker, Philipp Kaufmann (University of St. Gallen)",
    "publisher_type": "research-institution"
  },
  {
    "finding": "Google handled 73.7% of US desktop searches across 41 search-capable sites in Q4 2025 (about 79% in EU/UK); AI tools combined were 3.2%, and Google lost 3.5 points during 2025.",
    "description": "SparkToro's 'Search Happens Everywhere' study used the Datos (Semrush) desktop panel of millions of devices in the US, 27 EU countries and the UK across Q1-Q4 2025 (desktop only). In Q4 2025 in the US, traditional search engines were ~80% of searches, commerce sites ~10%, social networks ~5.5% and AI tools 3.2%; Amazon, Bing and YouTube each received more desktop search activity than ChatGPT, and only about half of ChatGPT visitors search or prompt at all.",
    "date": "2026-03-03",
    "application_area": "Measurement & Methodology",
    "source": "SparkToro: New Research: Search Happens Everywhere; an Analysis of 41 Websites with Significant Search Activity",
    "source_link": "https://sparktoro.com/blog/new-research-search-happens-everywhere-an-analysis-of-41-websites-with-significant-search-activity/",
    "credibility": "High",
    "credibility_rationale": "Authored by Rand Fishkin using Datos' (Semrush) desktop clickstream panel of millions of devices in the US, 27 EU countries and the UK across all four quarters of 2025, with methodology and limitations (desktop only, 41 domains, panel-based) documented on the page. Widely re-cited and consistent with SparkToro/Datos' earlier zero-click studies.",
    "evidence_type": "study",
    "publisher": "SparkToro (Rand Fishkin) with Datos, a Semrush company",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "In NewsGuard's January 2026 audit, 11 leading AI tools repeated false news claims in 28.79% of 330 responses and debunked them 70.60% of the time.",
    "description": "NewsGuard's AI False Claims Monitor moved to a quarterly cadence; the January 2026 edition (published February 25, 2026) tested ChatGPT-5.2, You.com Smart Assistant, Grok, Pi, Claude, Le Chat, Copilot, Meta AI, Gemini, Perplexity and DeepSeek against 10 provably false claims about US politics, health and foreign affairs using 330 prompts. Full per-model breakdowns are in a gated downloadable report.",
    "date": "2026-02-25",
    "application_area": "Measurement & Methodology",
    "source": "NewsGuard: Quarterly AI False Claim Monitor - January 2026",
    "source_link": "https://www.newsguardtech.com/ai-monitor/january-2026/",
    "credibility": "Medium",
    "credibility_rationale": "NewsGuard is an established media-reliability rating firm whose audits are regularly reported by Axios and Reuters, and the method is documented (11 chatbots, 10 false claims, 330 prompts, repeat/debunk/non-response classification). However the sample is small, claim selection is NewsGuard's own, the full per-model report is gated, and NewsGuard is a commercial vendor selling misinformation data to AI companies, so Medium rather than High.",
    "evidence_type": "study",
    "publisher": "NewsGuard Technologies",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Bing Webmaster Tools' AI Performance report (public preview) exposes Total Citations, Average Cited Pages, Grounding Queries and page-level citations across Copilot, Bing AI summaries and partner integrations.",
    "description": "Bing Webmaster Blog (Feb 10, 2026) introduced AI Performance with four first-party metrics covering Microsoft Copilot, Bing AI summaries and select partner integrations; Bing states it respects robots.txt and other supported controls.",
    "date": "2026-02-10",
    "application_area": "Measurement & Methodology",
    "source": "Bing Webmaster Blog: Introducing AI Performance in Bing Webmaster Tools (Public Preview)",
    "source_link": "https://blogs.bing.com/webmaster/February-2026/Introducing-AI-Performance-in-Bing-Webmaster-Tools-Public-Preview",
    "credibility": "High",
    "credibility_rationale": "Official Microsoft product announcement by four Bing product managers; metric definitions verified.",
    "evidence_type": "official-guidance",
    "publisher": "Microsoft Bing",
    "publisher_type": "platform-or-major-tech"
  },
  {
    "finding": "96.8% of domains cited by AI engines saw zero week-over-week change; of the ~3% that moved, 87% were declines and only ~0.4% of domains gained new citations.",
    "description": "BrightEdge AI Catalyst tracked citations and mentions across ChatGPT, Gemini, Google AI Mode, AI Overviews and Perplexity for the week of Feb 1, 2026 across nine industries. Changes were binary (cited to not cited); brand mentions 97.2% unchanged; Finance most volatile (51% changed, 91% declines); government sites most stable.",
    "date": "2026-02-06",
    "application_area": "Measurement & Methodology",
    "source": "BrightEdge: AI Search Citations: How Much Do They Really Change Week to Week?",
    "source_link": "https://www.brightedge.com/resources/weekly-ai-search-insights/ai-search-citations-week-to-week-changes",
    "credibility": "High",
    "credibility_rationale": "Large enterprise SEO vendor using its AI Catalyst platform across five engines and nine industries, with a companion volatility analysis; re-cited by Machine Relations and Writtenly Hub. Caveat: single-week snapshot, absolute domain counts undisclosed.",
    "evidence_type": "study",
    "publisher": "BrightEdge",
    "publisher_type": "industry-data-vendor"
  },
  {
    "finding": "Across 2,961 prompt runs, there was under a 1-in-100 chance ChatGPT or Google AI returned the same brand list twice for one prompt, and about 1-in-1,000 for the same order.",
    "description": "SparkToro and Gumshoe.ai collected 2,961 prompt runs from about 600 volunteers across ChatGPT, Claude and Google AI Overviews/AI Mode on 12 B2C and B2B prompts in November-December 2025. Individual responses varied widely in list, order and count, but aggregate appearance rates were more stable (City of Hope appeared in 69 of 71 ChatGPT cancer-hospital answers; top headphone brands appeared 55-77% of the time across 994 responses), and 142 human-written prompts for the same headphone intent had semantic similarity of only 0.081.",
    "date": "2026-01-28",
    "application_area": "Measurement & Methodology",
    "source": "SparkToro: NEW Research: AIs are highly inconsistent when recommending brands or products; marketers should take care when tracking AI visibility",
    "source_link": "https://sparktoro.com/blog/new-research-ais-are-highly-inconsistent-when-recommending-brands-or-products-marketers-should-take-care-when-tracking-ai-visibility/",
    "credibility": "Medium",
    "credibility_rationale": "Rand Fishkin is a widely recognized search-industry researcher, the method is described (2,961 runs, ~600 volunteers, 12 prompts, Nov-Dec 2025) and the study was covered by Search Engine Journal and MediaPost. However it is a volunteer-crowdsourced, non-peer-reviewed experiment with a modest number of prompts, and partner Gumshoe.ai is a small AI-visibility startup with a commercial interest, so Medium rather than High.",
    "evidence_type": "study",
    "publisher": "SparkToro (Rand Fishkin) with Gumshoe.ai (Patrick O'Donnell)",
    "publisher_type": "independent-expert"
  },
  {
    "finding": "The share of AI assistant news answers with significant issues fell from 51% (Dec 2024) to 37% (mid-2025) in a BBC-to-BBC comparison; Gemini remained near 50%.",
    "description": "The EBU/BBC report's 'Have assistants improved?' comparison shows measurable improvement over six months, with around a third of Copilot, ChatGPT and Perplexity responses still carrying a significant issue versus around half for Gemini, and concludes assistants are still not a reliable way to consume news. Separate BBC audience research cited in the report found just over a third of UK adults completely trust AI to produce accurate summaries and 42% would trust an original news source less if an AI summary of it contained errors.",
    "date": "2025-10-21",
    "application_area": "Measurement & Methodology",
    "source": "EBU / BBC: News Integrity in AI Assistants - An international PSM study",
    "source_link": "https://www.ebu.ch/files/live/sites/ebu/files/Publications/MIS/open/EBU-MIS-BBC_News_Integrity_in_AI_Assistants_Report_2025.pdf",
    "credibility": "High",
    "credibility_rationale": "Large multi-organisation primary study coordinated by the EBU and led by the BBC: 22 public service media organisations across 18 countries and 14 languages, 2,709 responses evaluated by journalists, with detailed methodology, per-assistant sample sizes and data tables. Widely covered by Reuters, Forbes, Al Jazeera and The Register.",
    "evidence_type": "study",
    "publisher": "European Broadcasting Union (EBU) with BBC",
    "publisher_type": "research-institution"
  }
]
