{
  "title": "AI progress in the 2021 report",
  "report_year": 2021,
  "period": "Historical snapshot: October 2021. Dates and populations are specified per row.",
  "methodology": "Training size is not a quality score. TruthfulQA was designed to expose particular failure modes; its rates should not be treated as universal accuracy measures.",
  "source_deck": "https://docs.google.com/presentation/d/1bwJDRC777rAf00Drthi9yT2c9b0MabWO5ZlksfvFzx8/edit?usp=sharing",
  "web_edition_date": "2026-10-11",
  "rows": [
    {
      "measure": "CLIP training pairs",
      "value": "400M",
      "definition": "Text-image pairs used for pretraining.",
      "printed_slide": 38,
      "pdf_page": 38,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2021.pdf#page=38"
    },
    {
      "measure": "Best model truthfulness",
      "value": "58%",
      "definition": "TruthfulQA comparison reported in the 2021 deck.",
      "printed_slide": 44,
      "pdf_page": 44,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2021.pdf#page=44"
    },
    {
      "measure": "Human baseline truthfulness",
      "value": "94%",
      "definition": "Human comparison on the same benchmark.",
      "printed_slide": 44,
      "pdf_page": 44,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2021.pdf#page=44"
    }
  ]
}
