{
  "generatedAt": "2026-07-03T00:00:00.000Z",
  "disclaimer": "Benchmark records are manually curated from public source pages. Missing scores are intentionally left blank and must not be inferred.",
  "benchmarks": [
    {
      "id": "mmlu",
      "label": "MMLU",
      "category": "knowledge",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "gpqa",
      "label": "GPQA",
      "category": "reasoning",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "swe_bench",
      "label": "SWE-bench",
      "category": "coding",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "human_eval",
      "label": "HumanEval",
      "category": "coding",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "aime",
      "label": "AIME",
      "category": "math",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "math",
      "label": "MATH",
      "category": "math",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "livebench",
      "label": "LiveBench",
      "category": "general",
      "higherIsBetter": true,
      "unit": "score"
    },
    {
      "id": "simplebench",
      "label": "SimpleBench",
      "category": "reasoning",
      "higherIsBetter": true,
      "unit": "percent"
    },
    {
      "id": "arena_elo",
      "label": "Arena Elo",
      "category": "preference",
      "higherIsBetter": true,
      "unit": "elo"
    }
  ],
  "records": [
    {
      "modelId": "openai-gpt-4-1-nano",
      "benchmarkId": "mmlu",
      "score": 80.1,
      "date": "2025-04-14",
      "source": "OpenAI",
      "sourceUrl": "https://openai.com/index/gpt-4-1/",
      "notes": "Reported by OpenAI for GPT-4.1 nano."
    },
    {
      "modelId": "openai-gpt-4-1-nano",
      "benchmarkId": "gpqa",
      "score": 50.3,
      "date": "2025-04-14",
      "source": "OpenAI",
      "sourceUrl": "https://openai.com/index/gpt-4-1/",
      "notes": "Reported by OpenAI for GPT-4.1 nano."
    },
    {
      "modelId": "openai-gpt-4o",
      "benchmarkId": "mmlu",
      "score": 88.7,
      "date": "2024-05-13",
      "source": "OpenAI GPT-4o release benchmark chart",
      "sourceUrl": "https://openai.com/index/hello-gpt-4o/",
      "notes": "Public GPT-4o benchmark score."
    },
    {
      "modelId": "openai-gpt-4o",
      "benchmarkId": "gpqa",
      "score": 53.6,
      "date": "2024-05-13",
      "source": "OpenAI GPT-4o release benchmark chart",
      "sourceUrl": "https://openai.com/index/hello-gpt-4o/",
      "notes": "Public GPT-4o benchmark score."
    },
    {
      "modelId": "openai-gpt-4o",
      "benchmarkId": "math",
      "score": 76.6,
      "date": "2024-05-13",
      "source": "OpenAI GPT-4o release benchmark chart",
      "sourceUrl": "https://openai.com/index/hello-gpt-4o/",
      "notes": "Public GPT-4o benchmark score."
    },
    {
      "modelId": "openai-gpt-4o",
      "benchmarkId": "human_eval",
      "score": 90.2,
      "date": "2024-05-13",
      "source": "OpenAI GPT-4o release benchmark chart",
      "sourceUrl": "https://openai.com/index/hello-gpt-4o/",
      "notes": "Public GPT-4o benchmark score."
    },
    {
      "modelId": "anthropic-claude-3-opus-20240229",
      "benchmarkId": "mmlu",
      "score": 86.8,
      "date": "2024-03-04",
      "source": "Anthropic Claude 3 family announcement",
      "sourceUrl": "https://www.anthropic.com/news/claude-3-family",
      "notes": "Reported by Anthropic for Claude 3 Opus."
    },
    {
      "modelId": "anthropic-claude-3-opus-20240229",
      "benchmarkId": "gpqa",
      "score": 50.4,
      "date": "2024-03-04",
      "source": "Anthropic Claude 3 family announcement",
      "sourceUrl": "https://www.anthropic.com/news/claude-3-family",
      "notes": "Reported by Anthropic for Claude 3 Opus."
    }
  ]
}
