{
  "generated": "2026-09-08",
  "benchmark": "mmmu-pro",
  "results": [
    {
      "system": "GPT-4o (0513)",
      "developer": "OpenAI",
      "value": 51.9,
      "date": "2024-09-04",
      "source": {
        "url": "https://arxiv.org/abs/2409.02813",
        "kind": "paper",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "notes": "Paper Table 1 overall (average of standard 10-option 54.0 and vision 49.7)."
      }
    },
    {
      "system": "o3",
      "developer": "OpenAI",
      "value": 76.4,
      "date": "2025-04-16",
      "source": {
        "url": "https://mmmu-benchmark.github.io/",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "notes": "Self-reported entry on the official MMMU leaderboard (source: author). Official leaderboard overall; OpenAI's GPT-5 post reports the same 76.4 (average of standard and vision)."
      }
    },
    {
      "system": "Gemini 3.0 Pro",
      "developer": "Google DeepMind",
      "value": 81.0,
      "date": "2025-11-18",
      "source": {
        "url": "https://mmmu-benchmark.github.io/",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "tools": false,
        "notes": "Self-reported entry on the official MMMU leaderboard (source: author). Google's Gemini 3 launch post reports the same 81%."
      }
    },
    {
      "system": "GPT-5.4 Thinking w/ tools",
      "developer": "OpenAI",
      "value": 82.1,
      "date": "2026-03-05",
      "source": {
        "url": "https://mmmu-benchmark.github.io/",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "tools": true,
        "notes": "Self-reported entry on the official MMMU leaderboard (source: author). Without tools the leaderboard lists 81.2."
      }
    },
    {
      "system": "Chance Vision 1.5",
      "developer": "Chance (chance.vision)",
      "value": 86.9,
      "date": "2026-07-01",
      "source": {
        "url": "https://mmmu-benchmark.github.io/",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "notes": "Self-reported entry on the official MMMU leaderboard (source: author). Top of the official MMMU-Pro leaderboard, above the 85.4 estimated high-expert level; leaderboard shows only month, so the 1st is used. benchlm.ai instead lists GPT-5.4 Pro at 94% from a vendor chart."
      }
    }
  ]
}
