{
  "generated": "2026-09-08",
  "benchmark": "gsm8k",
  "results": [
    {
      "system": "GPT-3 6B (finetuned)",
      "developer": "OpenAI",
      "value": 20.6,
      "date": "2021-10-27",
      "source": {
        "url": "https://arxiv.org/abs/2110.14168",
        "kind": "paper",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "notes": "Section 4.1 of the GSM8K paper: a 6B model finetuned with full natural-language solutions reaches 20.6% test accuracy (5.2% when outputting the answer directly). Larger 175B results are given only in figures."
      }
    },
    {
      "system": "DeepSeek-V4-Pro-Base",
      "developer": "DeepSeek",
      "value": 92.6,
      "date": "2026-04-22",
      "source": {
        "url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "shots": 8,
        "notes": "DeepSeek-V4 model card, base-model table (HF repo created 2026-04-22). Base model, exact match; chat models are no longer reported on GSM8K by most vendors."
      }
    }
  ]
}
