{
  "generated": "2026-09-08",
  "benchmark": "mbpp",
  "results": [
    {
      "system": "137B LM (few-shot)",
      "developer": "Google Research",
      "value": 59.6,
      "date": "2021-08-16",
      "source": {
        "url": "https://arxiv.org/abs/2108.07732",
        "kind": "paper",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "notes": "Paper abstract: the largest 137B model synthesizes solutions to 59.6% of MBPP problems with few-shot prompting; fine-tuning adds about 10 points."
      }
    },
    {
      "system": "Llama 3.1 405B Instruct",
      "developer": "Meta",
      "value": 88.6,
      "date": "2024-07-23",
      "source": {
        "url": "https://huggingface.co/meta-llama/Llama-3.1-70B-Instruct",
        "kind": "developer-report",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "pass_k": 1,
        "shots": 0,
        "notes": "Llama 3.1 model card row 'MBPP ++ base version' (EvalPlus MBPP+ base tests, 378 problems): 88.6 for 405B, 86.0 for 70B. Not the original 500-problem test split. Date is the Llama 3.1 release."
      }
    }
  ]
}
