{
  "about": "A read of 'GEO: Generative Engine Optimization' (Aggarwal et al., KDD '24). Figures read from the source PDF, not from secondary summaries.",
  "source_paper": {
    "title": "GEO: Generative Engine Optimization",
    "arxiv": "2311.09735",
    "version": "v3",
    "url": "https://arxiv.org/abs/2311.09735",
    "authors": [
      "Pranjal Aggarwal",
      "Vishvak Murahari",
      "Tanmay Rajpurohit",
      "Ashwin Kalyan",
      "Karthik Narasimhan",
      "Ameet Deshpande"
    ],
    "submitted": "2023-11-16",
    "revised": "2024-06-28",
    "venue": "KDD '24, Barcelona, 25-29 Aug 2024",
    "peer_reviewed": true,
    "funding": "US National Science Foundation, Grant No. 2107048",
    "verified": "2026-08-30"
  },
  "license": "CC BY 4.0",
  "canonical_url": "https://www.bestseopodcast.com/geo-paper-explained",
  "baseline_pawc": 19.5,
  "baseline_note": "Impressions are normalised so all citations in one response sum to 1, over the top 5 retrieved sources; an unoptimised source therefore sits near 1/5. The metric is a share of one response, so it is zero-sum by construction.",
  "methods": [
    {
      "name": "No Optimization",
      "group": "baseline",
      "pawc": 19.5,
      "subjective": 19.3,
      "note": "The control. Its value is ~1/5 because five sources share a total impression of 1.",
      "relative_pct_vs_baseline": 0.0
    },
    {
      "name": "Keyword Stuffing",
      "group": "non-performing",
      "pawc": 17.8,
      "subjective": 20.2,
      "note": "The single most useful result in the paper for practitioners, and almost never quoted: the classic SEO tactic measurably REDUCED visibility.",
      "relative_pct_vs_baseline": -8.7
    },
    {
      "name": "Unique Words",
      "group": "non-performing",
      "pawc": 20.7,
      "subjective": 20.4,
      "note": "Statistically close to baseline; the paper groups it as non-performing.",
      "relative_pct_vs_baseline": 6.2
    },
    {
      "name": "Easy-to-Understand",
      "group": "high-performing",
      "pawc": 22.2,
      "subjective": 20.5,
      "relative_pct_vs_baseline": 13.8
    },
    {
      "name": "Authoritative",
      "group": "high-performing",
      "pawc": 21.8,
      "subjective": 22.9,
      "note": "Best on debate-style and 'historical' domain queries.",
      "relative_pct_vs_baseline": 11.8
    },
    {
      "name": "Technical Terms",
      "group": "high-performing",
      "pawc": 23.1,
      "subjective": 21.4,
      "relative_pct_vs_baseline": 18.5
    },
    {
      "name": "Fluency Optimization",
      "group": "high-performing",
      "pawc": 25.1,
      "subjective": 21.9,
      "note": "Best partner in combination — pairs well with every other method.",
      "relative_pct_vs_baseline": 28.7
    },
    {
      "name": "Cite Sources",
      "group": "high-performing",
      "pawc": 24.9,
      "subjective": 21.9,
      "note": "The row behind the industry's \"+24.6%\". 24.6 is this row's Subjective Impression *Overall* sub-column — an absolute score, not a lift. Strongest on factual questions.",
      "relative_pct_vs_baseline": 27.7
    },
    {
      "name": "Statistics Addition",
      "group": "high-performing",
      "pawc": 25.9,
      "subjective": 23.7,
      "note": "Strongest in 'Law & Government' and 'Opinion' query types.",
      "relative_pct_vs_baseline": 32.8
    },
    {
      "name": "Quotation Addition",
      "group": "high-performing",
      "pawc": 27.8,
      "subjective": 24.7,
      "note": "The actual best performer, and the source of the \"up to 40%\" headline: (27.8-19.5)/19.5 = 42.6%, which the paper rounds to 41%. Best in 'People & Society', 'Explanation' and 'History'.",
      "relative_pct_vs_baseline": 42.6
    }
  ],
  "common_misquotes": [
    {
      "industry_claim": "Citing sources gives you a +24.6% visibility lift.",
      "verdict": "category-error",
      "what_paper_says": "24.6 is an ABSOLUTE score in Table 1 — captioned \"Absolute impression metrics\" — on a scale where doing nothing scores 19.5. It is not a lift. Cite Sources' real relative improvement on Position-Adjusted Word Count is 27.7%."
    },
    {
      "industry_claim": "GEO delivers up to a 40% visibility boost.",
      "verdict": "true-but-stripped-of-context",
      "what_paper_says": "That is the single best method (Quotation Addition) on the single best metric, averaged over the whole benchmark. On Perplexity.ai — the only commercially deployed engine tested — the figure is up to 37%. Most methods land between 12% and 33%, and one is negative."
    },
    {
      "industry_claim": "Tested across six generative engines.",
      "verdict": false,
      "what_paper_says": "The Limitations section states: \"we rigorously test our proposed methods on two generative engines, including a publicly available one.\" Two, not six. (Six appears in a different, later paper — arXiv 2603.29979 — which is a preprint, not peer reviewed.)"
    },
    {
      "industry_claim": "Proof that structure and citations drive AI visibility.",
      "verdict": "overstated",
      "what_paper_says": "Every subjective sub-metric was scored by GPT-3.5 using a G-Eval-style template, and GPT-3.5-turbo generated all the responses being scored. The headline is substantially a language model rating how prominently a language model cited a source. That is a legitimate and standard method, but it is not human-validated visibility, and no deck that quotes the number mentions it."
    }
  ],
  "buried_findings": [
    {
      "id": "zero-sum",
      "headline": "GEO visibility is zero-sum by construction.",
      "detail": "The paper normalises impressions \"so that the sum of the impressions of all citations in a response equals 1\", across the top 5 retrieved sources. Every point one source gains is a point the other four lose. This is not a caveat someone added later — it is how the metric is defined, and it means \"everyone can raise their AI visibility\" cannot be true as stated."
    },
    {
      "id": "rank-inversion",
      "headline": "The gains go to whoever is losing — and the leader pays for them.",
      "detail": "Section 5.2 tested what happens when every source optimises at once. Cite Sources produced a 115.1% visibility increase for the site ranked 5th, while \"on average, the visibility of the top-ranked website decreased by 30.3%.\" The authors read this as democratising. For a client already ranking first, it is a warning: it is the incumbent's share that gets redistributed."
    },
    {
      "id": "no-new-information",
      "headline": "The paper's own examples add persuasion, not substance.",
      "detail": "Table 4's caption states the methods work \"Without adding any substantial new information.\" Its worked example for Cite Sources scores +132.4% by attributing a statistic to \"The International Chocolate Consumption Research Group\" — a body that does not appear to exist. The authors present this as a demonstration of the method. It is also a demonstration of how the method rewards confident-sounding attribution over verified fact."
    },
    {
      "id": "keyword-stuffing-hurts",
      "headline": "Keyword stuffing measurably hurt.",
      "detail": "17.8 against a 19.5 baseline — an 8.7% decline, the only method to go backwards. Directly useful, and absent from every summary of this paper we could find."
    },
    {
      "id": "gpt35-era",
      "headline": "Everything here was measured on GPT-3.5 in 2023.",
      "detail": "Responses were generated by gpt-3.5-turbo at temperature 0.7, five samples, five random seeds, with the top 5 Google results as sources. The paper is careful and reproducible. But the engines it describes have been replaced wholesale since, and the authors say so: methods \"may need to adapt over time as GEs evolve.\""
    }
  ],
  "benchmark": {
    "name": "GEO-bench",
    "queries": 10000,
    "domains": 25,
    "query_types": 9,
    "categorisations": 7,
    "sources_per_query": 5,
    "generator_model": "gpt-3.5-turbo",
    "temperature": 0.7,
    "samples_per_query": 5,
    "random_seeds": 5,
    "scorer": "GPT-3.5, G-Eval-style template",
    "real_world_engine": "Perplexity.ai",
    "real_world_best_result_pct": 37
  },
  "takeaways": [
    "Quotation Addition and Statistics Addition beat Cite Sources. The industry quotes the third-best method.",
    "Keyword stuffing is the one tactic the paper shows going backwards. Stop doing it.",
    "Match the method to the query type: Authoritative for debate and history, Statistics for law/government and opinion, Cite Sources for factual, Quotation for people and society.",
    "Fluency Optimization is the best combiner — pair it rather than running it alone.",
    "If you already rank first, read Section 5.2 before adopting this wholesale: the modelled gains come out of the leader's share.",
    "Treat the magnitudes as directional and dated. The direction is credible and replicated elsewhere; the specific percentages are GPT-3.5-era artefacts."
  ]
}