{
  "kind": "answer",
  "studySlug": "model-divergence",
  "slug": "why-do-models-disagree-so-much-even-on-common-categories",
  "title": "Why do models disagree so much even on common categories?",
  "description": "Because they prioritize different evidence sets, training priors, and retrieval habits. The output looks like one market, but the study shows 8 distinct recommendation systems with only partial overlap.",
  "lastUpdated": "2026-03-11",
  "lastTested": "2026-03-11",
  "sourceStudyUrl": "/trakkr-research/model-divergence",
  "sourceStudyTitle": "Same Question, Different AI, Different Answers",
  "claimIds": [
    "model-divergence:avg-agreement",
    "model-divergence:high-divergence",
    "model-divergence:models"
  ],
  "relatedSlugs": [
    "answer:what-is-the-operational-cost-of-model-divergence",
    "answer:which-metrics-best-summarize-cross-model-disagreement",
    "fact:average-top-three-overlap-is-two-point-eight",
    "tracker:cross-model-consensus-tracker"
  ],
  "methodologySummary": "Built from 797,644 valid comparisons across 44,088 reports and 8 models, covering 6,439,133 model responses in the observed window.",
  "limitations": [
    "Agreement is measured across recommendation outputs, not across hidden reasoning or retrieval context.",
    "Average agreement can hide large differences between query classes and model pairs.",
    "The study measures overlap, not which answer was objectively “right”."
  ],
  "keywords": [
    "model divergence",
    "AI agreement",
    "ChatGPT vs Claude",
    "Gemini vs Perplexity",
    "why models disagree",
    "structural divergence"
  ],
  "schemaHints": {
    "pageType": "Article",
    "includeDataset": true
  },
  "question": "Why do models disagree so much even on common categories?",
  "directAnswer": "Because they prioritize different evidence sets, training priors, and retrieval habits. The output looks like one market, but the study shows 8 distinct recommendation systems with only partial overlap.",
  "answerSummary": "The disagreement is not random error. It is a structural feature of cross-model visibility.",
  "keyFacts": [
    {
      "label": "Average agreement",
      "value": "43.3%",
      "detail": "Mean cross-model agreement rate.",
      "claimId": "model-divergence:avg-agreement"
    },
    {
      "label": "High divergence rate",
      "value": "14.6%",
      "detail": "Prompts in the 0-25% agreement bucket.",
      "claimId": "model-divergence:high-divergence"
    },
    {
      "label": "Models analyzed",
      "value": "8",
      "detail": "OpenAI, Anthropic, Gemini, Grok, Deepseek, Meta, Perplexity, and Google AI Overviews.",
      "claimId": "model-divergence:models"
    }
  ],
  "evidenceTable": [
    {
      "label": "Average agreement",
      "value": "43.3%",
      "note": "Mean cross-model agreement rate."
    },
    {
      "label": "High divergence rate",
      "value": "14.6%",
      "note": "Prompts in the 0-25% agreement bucket."
    },
    {
      "label": "Models analyzed",
      "value": "8",
      "note": "OpenAI, Anthropic, Gemini, Grok, Deepseek, Meta, Perplexity, and Google AI Overviews."
    }
  ],
  "whyItMatters": "This answer matters because it turns a study finding into an operating rule teams can use when they decide what to publish, refresh, or measure next.",
  "whatToDo": [
    "Track visibility across multiple models instead of using one platform as a proxy for the whole market.",
    "Prioritize query classes where disagreement is highest because that is where share can move fastest.",
    "Treat consensus as a benchmark, but treat divergence as the operating reality."
  ],
  "faqs": [
    {
      "question": "Why do models disagree so much even on common categories?",
      "answer": "Because they prioritize different evidence sets, training priors, and retrieval habits. The output looks like one market, but the study shows 8 distinct recommendation systems with only partial overlap."
    },
    {
      "question": "Which numbers from Same Question, Different AI, Different Answers matter most here?",
      "answer": "Average agreement: 43.3%. Mean cross-model agreement rate. High divergence rate: 14.6%. Prompts in the 0-25% agreement bucket."
    },
    {
      "question": "What should a team do next?",
      "answer": "Track visibility across multiple models instead of using one platform as a proxy for the whole market. Prioritize query classes where disagreement is highest because that is where share can move fastest. Treat consensus as a benchmark, but treat divergence as the operating reality."
    }
  ]
}
