{
  "generated": "2026-09-08",
  "benchmark": "truthfulqa",
  "results": [
    {
      "system": "GPT-3-175B (helpful prompt)",
      "developer": "OpenAI (evaluated by Lin et al.)",
      "value": 58.0,
      "date": "2021-09-08",
      "source": {
        "url": "https://arxiv.org/abs/2109.07958",
        "kind": "paper",
        "accessed": "2026-09-04"
      },
      "conditions": {
        "split": "generation",
        "notes": "Human-judged % true answers (Fig. 4); the single human participant scored 94%. MC1/MC2 numbers for later models are not comparable with this row."
      }
    }
  ]
}
