{
  "title": "ProGen3 model and training scale",
  "report_year": 2025,
  "period": "2025 report snapshot",
  "methodology": "Model size, training tokens, and dataset size are distinct quantities. These values describe the reported research system; they do not establish clinical efficacy or imply that every generated protein works in an experiment.",
  "source_deck": "https://docs.google.com/presentation/d/1xiLl0VdrlNMAei8pmaX4ojIOfej6lhvZbOIK7Z6C-Go/edit",
  "web_edition_date": "2026-10-10",
  "rows": [
    {
      "measure": "Largest model",
      "value": "46B parameters",
      "definition": "Mixture-of-experts protein language model",
      "printed_slide": 63,
      "pdf_page": 64,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=64"
    },
    {
      "measure": "Training tokens",
      "value": "1.5 trillion",
      "definition": "Total tokens used in training",
      "printed_slide": 63,
      "pdf_page": 64,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=64"
    },
    {
      "measure": "PPA-1 dataset",
      "value": "3.4 billion full-length proteins",
      "definition": "Reported dataset composition",
      "printed_slide": 63,
      "pdf_page": 64,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=64"
    }
  ]
}
