{
  "title": "AI safety in the 2024 report",
  "report_year": 2024,
  "period": "Historical snapshot: October 2024. Dates and populations are specified per row.",
  "methodology": "These rows identify methods, models, and institutions rather than comparable safety scores. The report did not claim that jailbreak defenses were universal or that interpreting selected features explained all model behavior.",
  "source_deck": "https://docs.google.com/presentation/d/1GmZmoWOa2O92BPrncRcTKa15xvQGhq7g4I4hJSNlC0M/edit",
  "web_edition_date": "2026-10-10",
  "rows": [
    {
      "measure": "Instruction hierarchy",
      "value": "Deployed in GPT-4o Mini",
      "definition": "Defense described as prioritizing instructions rather than treating all text as equally authoritative.",
      "printed_slide": 181,
      "pdf_page": 182,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2024.pdf#page=182"
    },
    {
      "measure": "Interpretability model studied",
      "value": "Claude 3 Sonnet",
      "definition": "Model whose activations were decomposed into interpretable features in the cited Anthropic work.",
      "printed_slide": 197,
      "pdf_page": 198,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2024.pdf#page=198"
    },
    {
      "measure": "Public evaluation framework",
      "value": "Inspect",
      "definition": "Framework released by the UK AI Safety Institute.",
      "printed_slide": 178,
      "pdf_page": 179,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2024.pdf#page=179"
    }
  ]
}
