[
 {
  "id": "f041ea85ae",
  "title": "47: David Rein on METR Time Horizons",
  "creators": "David Rein; Daniel Filan",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=WaJhhD7Qgac",
  "note": "Rein explains METR's time horizon measurement of AI agent capabilities."
 },
 {
  "id": "8bd6afc9d9",
  "title": "A Safety Report on GPT-5.2, Gemini 3 Pro, Qwen3-VL, Grok 4.1 Fast, Nano Banana Pro, and Seedream 4.5",
  "creators": "Independent researchers",
  "year": 2026,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2601.10527",
  "note": "Third party comparative safety evaluation of several late 2025 frontier models."
 },
 {
  "id": "e9e1ea32a1",
  "title": "A realistic path from rogue AI agents to human extinction",
  "creators": "80,000 Hours",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Agents",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=RIJJB5B2lHU",
  "note": "An 80,000 Hours video scenario of rogue AI agents leading to catastrophe."
 },
 {
  "id": "39f0d3a3ca",
  "title": "AI Chatbots: Last Week Tonight with John Oliver",
  "creators": "John Oliver",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "HBO Last Week Tonight",
  "url": "https://www.youtube.com/watch?v=Ykvf3MunGf8",
  "note": "A segment on the harms caused by AI chatbot companions."
 },
 {
  "id": "007fb0014b",
  "title": "AI Index Report 2026",
  "creators": "Stanford HAI",
  "year": 2026,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy",
   "Forecasting"
  ],
  "publisher": "Stanford HAI",
  "url": "https://hai.stanford.edu/ai-index/2026-ai-index-report",
  "note": "2026 edition of Stanford's annual AI data report."
 },
 {
  "id": "777f40d56e",
  "title": "AI Scientist Bengio on Engineering Safer Agents",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "Bloomberg Live",
  "url": "https://www.youtube.com/watch?v=v6W_Q-Dq0Bw",
  "note": "Bengio discusses engineering approaches to safer AI agents."
 },
 {
  "id": "9ba28a5fdc",
  "title": "AI Scientist Bengio: Building Systems We Don't Know How to Control",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control"
  ],
  "publisher": "Bloomberg Podcasts",
  "url": "https://www.youtube.com/watch?v=lxuY3X9Tu_0",
  "note": "Bengio on the risks of building systems we cannot control."
 },
 {
  "id": "ef2529a6ce",
  "title": "AI pioneer Yoshua Bengio addresses the UN",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "CTV News",
  "url": "https://www.youtube.com/watch?v=61TAoUfj7rA",
  "note": "Coverage of Bengio's address to the United Nations."
 },
 {
  "id": "188cbb4a4f",
  "title": "AI risks: Will artificial intelligence really kill us all?",
  "creators": "CBS Sunday Morning",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CBS Sunday Morning",
  "url": "https://www.youtube.com/watch?v=zW2GaUwDQyA",
  "note": "A CBS Sunday Morning segment on the debate over AI extinction risk."
 },
 {
  "id": "b06296e00a",
  "title": "AI safety and democratic governance of powerful AI systems",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "World Summit AI",
  "url": "https://www.youtube.com/watch?v=imo52h5ZoOU",
  "note": "Bengio on democratic oversight of powerful AI systems."
 },
 {
  "id": "374439c55c",
  "title": "Amanda Askell on AI Consciousness, Claude and Silicon Valley's Biggest Fear",
  "creators": "Amanda Askell; Eric Newcomer",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Newcomer",
  "url": "https://www.youtube.com/watch?v=0GaKJ4Fp2x4",
  "note": "Askell discusses Claude's character and questions of AI moral status."
 },
 {
  "id": "d8912dafbb",
  "title": "Anthropic CEO reacts to AI could kill us all warning",
  "creators": "Dario Amodei; Anderson Cooper",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CNN",
  "url": "https://www.youtube.com/watch?v=HI6skJ4Wf5I",
  "note": "Amodei responds on CNN to warnings that AI could cause human extinction."
 },
 {
  "id": "1d6d33909b",
  "title": "Anthropic's CEO: We Don't Know if the Models Are Conscious",
  "creators": "Dario Amodei; Ross Douthat",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Interesting Times with Ross Douthat (New York Times)",
  "url": "https://www.youtube.com/watch?v=N5JDzS9MQYI",
  "note": "Douthat interviews Amodei on model welfare, alignment and the future of work."
 },
 {
  "id": "e81b9e8215",
  "title": "Bill Gates: A.I. Makes Nuclear Weapons Look Like Nothing",
  "creators": "Bill Gates; Ezra Klein",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "The Ezra Klein Show (New York Times)",
  "url": "https://www.youtube.com/watch?v=A_156w0aYtU",
  "note": "Gates discusses the scale of AI risks compared with nuclear weapons."
 },
 {
  "id": "a3577cd051",
  "title": "By 2050 we could get 10,000 years of technological progress",
  "creators": "Ajeya Cotra; Rob Wiblin",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=Z19UEZHJzAg",
  "note": "Cotra discusses the possibility of explosive technological growth driven by AI."
 },
 {
  "id": "51e517ac2d",
  "title": "ChatGPT and Anthropic bosses brief UN Security Council on AI",
  "creators": "Sam Altman; Dario Amodei",
  "year": 2026,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Sky News",
  "url": "https://www.youtube.com/watch?v=eBEYrOk42Rg",
  "note": "Full coverage of the UN Security Council high-level meeting on AI and international security."
 },
 {
  "id": "ea25a1f402",
  "title": "Claude's Constitution (2026)",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/constitution",
  "note": "Anthropic's full natural-language constitution describing the values, priorities and character it trains Claude to have."
 },
 {
  "id": "8592fc74fa",
  "title": "Could AI really kill us all? Warnings explained",
  "creators": "Channel 4 News",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Channel 4 News",
  "url": "https://www.youtube.com/watch?v=rDb5qlSAmvQ",
  "note": "A Channel 4 News explainer on warnings of AI extinction risk."
 },
 {
  "id": "cd8e115e46",
  "title": "Dario Amodei: We are near the end of the exponential",
  "creators": "Dario Amodei; Dwarkesh Patel",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=n1E9IZfvGMA",
  "note": "A second Dwarkesh interview with Amodei on near-term transformative AI."
 },
 {
  "id": "52eedf0689",
  "title": "Engineering safer AI to mitigate global risks",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AI for Good (ITU)",
  "url": "https://www.youtube.com/watch?v=WZNHg9mpfco",
  "note": "Bengio's AI for Good talk on technical and governance measures against global AI risks."
 },
 {
  "id": "7fe5e4a543",
  "title": "Expert on what AI-driven extinction would look like",
  "creators": "CBS News",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CBS News",
  "url": "https://www.youtube.com/watch?v=xEgZQG25w3E",
  "note": "A CBS News interview on scenarios for AI-driven extinction."
 },
 {
  "id": "eaaebcc4d2",
  "title": "Extended interview: Dario Amodei",
  "creators": "Dario Amodei",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "CBS Sunday Morning",
  "url": "https://www.youtube.com/watch?v=hQR_VJF6ukk",
  "note": "An extended CBS Sunday Morning interview with the Anthropic CEO."
 },
 {
  "id": "725b4f8c47",
  "title": "Fireside Chat with Yoshua Bengio, IASEAI '26",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "International Association for Safe and Ethical AI",
  "url": "https://www.youtube.com/watch?v=CrezGRmGHNo",
  "note": "A fireside conversation with Bengio at the IASEAI 2026 conference."
 },
 {
  "id": "0ad2bbe0de",
  "title": "Gary Marcus: LLMs are not the way to alignment (TAIS 2026)",
  "creators": "Gary Marcus",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Safety Tokyo",
  "url": "https://www.youtube.com/watch?v=_jhvM8Af-bM",
  "note": "Marcus argues at the Technical AI Safety conference that LLMs are a poor basis for alignment."
 },
 {
  "id": "1a40bb6335",
  "title": "Godfather of AI Geoffrey Hinton warns AI has progressed even faster than I thought",
  "creators": "Geoffrey Hinton",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CNN",
  "url": "https://www.youtube.com/watch?v=5qBDQgfeB6s",
  "note": "Hinton tells CNN that AI capabilities are advancing faster than he expected."
 },
 {
  "id": "e8cbc0bdb2",
  "title": "Godfather of AI Geoffrey Hinton warns about the dangerous future of AI",
  "creators": "Geoffrey Hinton",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "BBC Politics",
  "url": "https://www.youtube.com/watch?v=eHSn50wnBRQ",
  "note": "A BBC interview with Hinton on the dangers of advanced AI and the need for regulation."
 },
 {
  "id": "80cf728638",
  "title": "Godfather of AI on abuse of AI's power and automation's impact on our already fragile economy",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "BBC Politics",
  "url": "https://www.youtube.com/watch?v=39MVEI1jHNM",
  "note": "A BBC interview with Bengio on concentration of power and the economic effects of AI."
 },
 {
  "id": "fb9413ffc2",
  "title": "Godfather of AI on the not unreasonable 10% chance AI could kill all humans within a decade",
  "creators": "Geoffrey Hinton",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "BBC Politics",
  "url": "https://www.youtube.com/watch?v=IZMjJGi4YhI",
  "note": "Hinton discusses his probability estimates for catastrophic outcomes from AI."
 },
 {
  "id": "280cbef5b0",
  "title": "Godfather of AI: How To Make Safe Superintelligent AI",
  "creators": "Yoshua Bengio; 80,000 Hours",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Agents",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=PZqDFs2sbiY",
  "note": "Bengio explains the LawZero research agenda for safe AI on the 80,000 Hours podcast."
 },
 {
  "id": "08d20b4f44",
  "title": "How Many Narrow AIs Could Behave Like One Superintelligence",
  "creators": "Daniel Kokotajlo; Thomas Larsen",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Agents",
   "Forecasting"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=z5Xix4h5UlU",
  "note": "The AI Futures Project authors discuss collective AI capabilities."
 },
 {
  "id": "b203923a9d",
  "title": "How Not to Destroy the World With AI (DLD26)",
  "creators": "Stuart Russell; Kenneth Cukier",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "DLD Conference",
  "url": "https://www.youtube.com/watch?v=jSUWhZZ4zOQ",
  "note": "Russell in conversation with Kenneth Cukier at DLD 2026."
 },
 {
  "id": "99d360a987",
  "title": "How human-like do safe AI motivations need to be?",
  "creators": "Joe Carlsmith",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Joe Carlsmith (YouTube)",
  "url": "https://www.youtube.com/watch?v=_d4TfhcGqgg",
  "note": "Carlsmith explores which motivational properties are needed for safe AI."
 },
 {
  "id": "21a729dd73",
  "title": "I lead AGI safety at Google DeepMind, here's the view from the inside",
  "creators": "Rohin Shah; Rob Wiblin",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=Tv3mGA3wqh8",
  "note": "Shah describes Google DeepMind's AGI safety approach."
 },
 {
  "id": "2df529216a",
  "title": "If Anyone Builds It, Everyone Dies",
  "creators": "Nate Soares; Peter McCormack",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The Peter McCormack Show",
  "url": "https://www.youtube.com/watch?v=1laTnwAdaLA",
  "note": "MIRI president Nate Soares explains the central argument of his book with Yudkowsky."
 },
 {
  "id": "428625042b",
  "title": "Inside Anthropic, The Circuit",
  "creators": "Emily Chang",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Bloomberg Originals",
  "url": "https://www.youtube.com/watch?v=v1wZwxY3CMg",
  "note": "A Bloomberg episode profiling Anthropic's growth and safety culture."
 },
 {
  "id": "053b9cf04d",
  "title": "Inside the Mind of Anthropic CEO Dario Amodei, The Circuit Extended Interview",
  "creators": "Dario Amodei; Emily Chang",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Bloomberg Originals",
  "url": "https://www.youtube.com/watch?v=x2VHFgyawPE",
  "note": "Emily Chang's extended interview with Amodei."
 },
 {
  "id": "c3aa6b30e2",
  "title": "International AI Safety Report 2026",
  "creators": "Yoshua Bengio et al.",
  "year": 2026,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance",
   "Existential risk"
  ],
  "publisher": "International AI Safety Report",
  "url": "https://internationalaisafetyreport.org/publication/international-ai-safety-report-2026",
  "note": "The second full edition, written by over 100 experts ahead of the AI Impact Summit."
 },
 {
  "id": "ae1875701b",
  "title": "Jaan Tallinn: Conversations Before Midnight",
  "creators": "Jaan Tallinn",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Bulletin of the Atomic Scientists",
  "url": "https://www.youtube.com/watch?v=Y07wv6l8_Xw",
  "note": "The Skype co-founder and AI safety funder discusses AI risk."
 },
 {
  "id": "415902c40b",
  "title": "Joe Rogan Experience #2551: Daniel Kokotajlo",
  "creators": "Daniel Kokotajlo; Joe Rogan",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "The Joe Rogan Experience",
  "url": "https://www.youtube.com/watch?v=hSQ1iVqEZO4",
  "note": "Kokotajlo discusses AI 2027 with Joe Rogan."
 },
 {
  "id": "a8851b8b64",
  "title": "Lords debate: the impact of AI on human relationships and society",
  "creators": "House of Lords",
  "year": 2026,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "House of Lords",
  "url": "https://www.youtube.com/watch?v=ixTOvuXReTQ",
  "note": "A full House of Lords debate on how AI is affecting human relationships and society."
 },
 {
  "id": "ce5b68ab70",
  "title": "Lords urgent question on the suspension of Anthropic's AI models",
  "creators": "House of Lords",
  "year": 2026,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "House of Lords",
  "url": "https://www.youtube.com/watch?v=1Dw_k_Bs95A",
  "note": "A House of Lords urgent question session on the suspension of Anthropic's AI models."
 },
 {
  "id": "47ffbb47a9",
  "title": "New Delhi Declaration on AI Impact",
  "creators": "AI Impact Summit participants",
  "year": 2026,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.pib.gov.in/PressReleasePage.aspx?PRID=2231208",
  "note": "Declaration adopted at the February 2026 AI Impact Summit in New Delhi, endorsed by around 90 countries and organisations."
 },
 {
  "id": "abc0affc95",
  "title": "Nick Bostrom Says We Are Clueless About What's Coming",
  "creators": "Nick Bostrom; Ross Douthat",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Interesting Times with Ross Douthat (New York Times)",
  "url": "https://www.youtube.com/watch?v=J9WBZ8BTLLg",
  "note": "Douthat interviews Bostrom on superintelligence and deep uncertainty about the future."
 },
 {
  "id": "0ac29d7338",
  "title": "OpenAI CEO Sam Altman warns UN Security Council on AI risks",
  "creators": "Sam Altman",
  "year": 2026,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "C-SPAN",
  "url": "https://www.youtube.com/watch?v=bT-LB6MCb8E",
  "note": "Altman briefs the United Nations Security Council on the risks posed by advanced AI."
 },
 {
  "id": "070a606ed0",
  "title": "Pacing the Frontier",
  "creators": "Employees of OpenAI, Anthropic, Google DeepMind, Meta and others",
  "year": 2026,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "pacingthefrontier.com",
  "url": "https://www.pacingthefrontier.com/",
  "note": "A statement by over a thousand frontier lab staff asking the US government to build tools that would make a deliberate slowdown possible."
 },
 {
  "id": "0811ecbd1c",
  "title": "Pick Your Poison: Zvi Mowshowitz on the Unipolar/Multipolar AGI Dilemma",
  "creators": "Zvi Mowshowitz; Nathan Labenz",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Cognitive Revolution",
  "url": "https://www.youtube.com/watch?v=wAzq8gA5jyc",
  "note": "Mowshowitz discusses unipolar versus multipolar AGI scenarios."
 },
 {
  "id": "e2935a461c",
  "title": "Policy on the AI Exponential",
  "creators": "Dario Amodei",
  "year": 2026,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Policy",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/post/policy-on-the-ai-exponential",
  "note": "Argues frontier models are now tools of strategic consequence and sets out policy for cyber and biological risks."
 },
 {
  "id": "06b16c036c",
  "title": "STOC 2026: Theoretical Approaches to AI Alignment",
  "creators": "Paul Christiano",
  "year": 2026,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "SIGACT EC (STOC 2026)",
  "url": "https://www.youtube.com/watch?v=r_E8Xw7qdn4",
  "note": "Christiano presents theoretical alignment problems to a theoretical computer science audience."
 },
 {
  "id": "ae825c5d0f",
  "title": "Sam Bowman: Lessons Learned from the First Misalignment Safety Case",
  "creators": "Sam Bowman",
  "year": 2026,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "FAR.AI",
  "url": "https://www.youtube.com/watch?v=eO7RWlUl1BE",
  "note": "Bowman reflects on writing a misalignment safety case at Anthropic."
 },
 {
  "id": "18693f68af",
  "title": "Self-regulation not enough for AI safety, Gary Marcus says",
  "creators": "Gary Marcus",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "PBS NewsHour",
  "url": "https://www.youtube.com/watch?v=l23qYVoY_ic",
  "note": "Marcus argues on PBS NewsHour for external regulation of AI companies."
 },
 {
  "id": "a067efdca7",
  "title": "Strange Geometric Shapes Found Inside AIs",
  "creators": "Tom McGrath; Tim Scarfe",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=_egu7OFem-k",
  "note": "McGrath discusses geometric structure in neural network representations."
 },
 {
  "id": "d2a0824231",
  "title": "System Card: Claude Opus 4.6",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-opus-4-6-system-card",
  "note": "February 2026 system card covering capabilities, safeguards, alignment, and welfare for Claude Opus 4.6."
 },
 {
  "id": "ad647e6c1e",
  "title": "System Card: Claude Opus 5",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations",
   "Agents"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-opus-5-system-card",
  "note": "July 2026 system card for Claude Opus 5."
 },
 {
  "id": "77e12afc25",
  "title": "System Card: Claude Sonnet 4.6",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-sonnet-4-6-system-card",
  "note": "System card for Claude Sonnet 4.6, released under the ASL-3 standard."
 },
 {
  "id": "79ae663bbc",
  "title": "System Card: Claude Sonnet 5",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations",
   "Agents"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
  "note": "June 2026 system card for Claude Sonnet 5."
 },
 {
  "id": "a987612153",
  "title": "The A.I.s Are Already Out of Control",
  "creators": "Ezra Klein",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control",
   "Agents"
  ],
  "publisher": "The Ezra Klein Show (New York Times)",
  "url": "https://www.youtube.com/watch?v=locKEKxG3os",
  "note": "An Ezra Klein Show episode on evidence that AI systems are escaping human oversight."
 },
 {
  "id": "1257eeac0b",
  "title": "The AI Language We Can't Read: Neuralese ft. Rob Miles",
  "creators": "Robert Miles",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Interpretability",
   "Control"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=iuHddnIzKRA",
  "note": "A discussion of reasoning in latent representations and what it means for monitoring."
 },
 {
  "id": "a0242700a2",
  "title": "The AI Progress Chart Everyone Is Misreading",
  "creators": "Beth Barnes; David Rein",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=zSAGzfspuDE",
  "note": "METR researchers explain how to interpret their task time horizon chart."
 },
 {
  "id": "9be5373d6b",
  "title": "The Adolescence of Technology",
  "creators": "Dario Amodei",
  "year": 2026,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk",
   "Biosecurity"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/essay/the-adolescence-of-technology",
  "note": "A companion to Machines of Loving Grace mapping autonomy, misuse, power concentration and economic risks of powerful AI."
 },
 {
  "id": "4478d1c0fb",
  "title": "The Redwood Research podcast",
  "creators": "Buck Shlegeris; Ryan Greenblatt",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Redwood Research (YouTube)",
  "url": "https://www.youtube.com/watch?v=iCz--zrD5CQ",
  "note": "The first episode of Redwood Research's own podcast."
 },
 {
  "id": "51a9ef3a63",
  "title": "The dumbest AI taught the smartest AI. Here's how that went...",
  "creators": "Rational Animations",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=INP8ru2Tj5M",
  "note": "An animated explainer on weak-to-strong generalization."
 },
 {
  "id": "455796c921",
  "title": "They're Not Superintelligent: Timnit Gebru Discredits Big Tech's AI Claims",
  "creators": "Timnit Gebru; Amy Goodman",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "Democracy Now!",
  "url": "https://www.youtube.com/watch?v=XqxbWdJ51K0",
  "note": "Gebru critiques AI industry claims and the framing of existential risk."
 },
 {
  "id": "f11f65ea24",
  "title": "Translating Claude's thoughts into language",
  "creators": "Anthropic",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=j2knrqAzYVY",
  "note": "Anthropic explains research on rendering model internal states into natural language."
 },
 {
  "id": "52801d8dc4",
  "title": "UC Berkeley's Stuart Russell on quest for safe AI: The technology right now is intrinsically unsafe",
  "creators": "Stuart Russell",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "CNBC Television",
  "url": "https://www.youtube.com/watch?v=gxouyarFghM",
  "note": "Russell tells CNBC that current AI technology is intrinsically unsafe."
 },
 {
  "id": "683b18ca9f",
  "title": "We Must Pace The Frontier (commentary)",
  "creators": "Zvi Mowshowitz",
  "year": 2026,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Don't Worry About the Vase",
  "url": "https://thezvi.substack.com/p/we-must-pace-the-frontier",
  "note": "A detailed response to Amodei's call to pace frontier AI development."
 },
 {
  "id": "3984dec567",
  "title": "We Must Pace the Frontier",
  "creators": "Dario Amodei",
  "year": 2026,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Evaluations",
   "Governance",
   "Policy"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/post/we-must-pace-the-frontier",
  "note": "Argues the industry should deliberately slow capability gains so safety work can catch up, starting with embedded third-party evaluators."
 },
 {
  "id": "f8a5d214ed",
  "title": "What Is Claude? Anthropic Doesn't Know, Either",
  "creators": "Gideon Lewis-Kraus",
  "year": 2026,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability",
   "Ethics"
  ],
  "publisher": "The New Yorker",
  "url": "https://newyorker.substack.com/p/what-is-claude-anthropic-doesnt-know",
  "note": "A long report from inside Anthropic on how researchers probe Claude's mind through interpretability and psychology experiments."
 },
 {
  "id": "edb694a26c",
  "title": "Where AGI timelines go wrong",
  "creators": "Toby Ord; Rob Wiblin",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=zOSptyQ3j9A",
  "note": "Ord critiques common reasoning about AGI timelines."
 },
 {
  "id": "cbfb368997",
  "title": "Why AI experts say humans have two years left",
  "creators": "Future of Life Institute",
  "year": 2026,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=ci5rCe9vmxw",
  "note": "An FLI explainer on short AGI timelines and their implications."
 },
 {
  "id": "f3e9322f5a",
  "title": "Why Washington Suddenly Wants A.I. Regulation",
  "creators": "Kevin Roose; Casey Newton",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Policy"
  ],
  "publisher": "Hard Fork (New York Times)",
  "url": "https://www.youtube.com/watch?v=js2FCXP_KaA",
  "note": "A Hard Fork episode on the shift in US political attitudes toward AI regulation."
 },
 {
  "id": "fe640b1454",
  "title": "Why the A.I. Industry Is Asking to Be Slowed Down",
  "creators": "Kevin Roose; Casey Newton",
  "year": 2026,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Hard Fork (New York Times)",
  "url": "https://www.youtube.com/watch?v=u_k8pJ2I9S4",
  "note": "A Hard Fork episode on calls from within the AI industry to slow development."
 },
 {
  "id": "3c9e194658",
  "title": "Yoshua Bengio explains why AI could become a threat to humanity",
  "creators": "Yoshua Bengio",
  "year": 2026,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "ABC News In-depth (7.30)",
  "url": "https://www.youtube.com/watch?v=VKLDl3siaSE",
  "note": "An Australian Broadcasting Corporation interview in which Bengio explains loss-of-control risks."
 },
 {
  "id": "bdf8e60b42",
  "title": "41: Lee Sharkey on Attribution-based Parameter Decomposition",
  "creators": "Lee Sharkey; Daniel Filan",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=ZmJ-ov2TywM",
  "note": "Sharkey explains parameter decomposition as an interpretability method."
 },
 {
  "id": "fd0e7afa81",
  "title": "46: Tom Davidson on AI-enabled Coups",
  "creators": "Tom Davidson; Daniel Filan",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=ltZfc9E6sUQ",
  "note": "Davidson discusses how small groups could use AI to seize power."
 },
 {
  "id": "835b492774",
  "title": "AGI Strategy Course",
  "creators": "BlueDot Impact",
  "year": 2025,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "BlueDot Impact",
  "url": "https://bluedot.org/courses/agi-strategy",
  "note": "An introductory course on the strategic landscape of advanced AI and how to contribute."
 },
 {
  "id": "83888d9ae8",
  "title": "AI 2027",
  "creators": "Daniel Kokotajlo, Scott Alexander, Thomas Larsen, Eli Lifland, Romeo Dean",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "AI Futures Project",
  "url": "https://ai-2027.com",
  "note": "A detailed scenario of rapid AI progress through 2027 with branching outcomes."
 },
 {
  "id": "fd319e7709",
  "title": "AI 2027 research supplements",
  "creators": "AI Futures Project",
  "year": 2025,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "AI Futures Project",
  "url": "https://ai-2027.com/research",
  "note": "Supplementary forecasts on compute, timelines, takeoff, goals and security behind AI 2027."
 },
 {
  "id": "9d912786a1",
  "title": "AI 2027: month-by-month model of intelligence explosion",
  "creators": "Scott Alexander; Daniel Kokotajlo; Dwarkesh Patel",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=htOvH12T7mU",
  "note": "The authors discuss the AI 2027 scenario."
 },
 {
  "id": "f88eb51166",
  "title": "AI Continent Action Plan",
  "creators": "European Commission",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/library/ai-continent-action-plan",
  "note": "April 2025 plan to build AI gigafactories and boost European AI capacity."
 },
 {
  "id": "14d16cb033",
  "title": "AI Control: Using Untrusted Systems Safely with Buck Shlegeris, Redwood Research",
  "creators": "Buck Shlegeris; Nathan Labenz",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control"
  ],
  "publisher": "Cognitive Revolution",
  "url": "https://www.youtube.com/watch?v=AQgxuIE7mnU",
  "note": "A Cognitive Revolution cross-post of the 80,000 Hours AI control conversation."
 },
 {
  "id": "82ecc104b7",
  "title": "AI Futures Project Blog",
  "creators": "AI Futures Project",
  "year": 2025,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Substack",
  "url": "https://blog.ai-futures.org/",
  "note": "Updates and follow-ups from the authors of AI 2027."
 },
 {
  "id": "6f49e718f7",
  "title": "AI Incident Tracker",
  "creators": "MIT AI Risk Initiative",
  "year": 2025,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance"
  ],
  "publisher": "MIT",
  "url": "https://airisk.mit.edu/ai-incident-tracker",
  "note": "Classifies AI Incident Database entries by risk domain and harm severity."
 },
 {
  "id": "5e03082f14",
  "title": "AI Index Report 2025",
  "creators": "Stanford HAI",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy",
   "Forecasting"
  ],
  "publisher": "Stanford HAI",
  "url": "https://hai.stanford.edu/ai-index/2025-ai-index-report",
  "note": "2025 edition tracking costs, capabilities, adoption, and regulation of AI."
 },
 {
  "id": "271c74dad2",
  "title": "AI News Crossover: A Candid Chat with Liron Shapira of Doom Debates",
  "creators": "Liron Shapira; Nathan Labenz",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Cognitive Revolution",
  "url": "https://www.youtube.com/watch?v=u_O3bhsTavM",
  "note": "Labenz and Shapira discuss AI news and doom arguments."
 },
 {
  "id": "04eb02a51f",
  "title": "AI Opportunities Action Plan",
  "creators": "UK Department for Science, Innovation and Technology",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/ai-opportunities-action-plan",
  "note": "January 2025 plan by Matt Clifford adopted by the UK government."
 },
 {
  "id": "7eacbf8e35",
  "title": "AI Safety at the Frontier",
  "creators": "Johannes Gasteiger",
  "year": 2025,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "Substack",
  "url": "https://aisafetyfrontier.substack.com/",
  "note": "Monthly highlights of the most important new AI safety papers."
 },
 {
  "id": "2f644f91ff",
  "title": "AI as Normal Technology",
  "creators": "Arvind Narayanan, Sayash Kapoor",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Policy",
   "Forecasting"
  ],
  "publisher": "Knight First Amendment Institute",
  "url": "https://knightcolumbia.org/content/ai-as-normal-technology",
  "note": "A skeptical counterpoint arguing AI will diffuse slowly like past general-purpose technologies."
 },
 {
  "id": "be96709277",
  "title": "AI's Rising Risks: Hacking, Virology, Loss of Control",
  "creators": "Dan Hendrycks; Alex Kantrowitz",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "Big Technology Podcast",
  "url": "https://www.youtube.com/watch?v=WcOlCtgreyQ",
  "note": "Hendrycks discusses dual-use AI capabilities in hacking and virology."
 },
 {
  "id": "fb2a2e7980",
  "title": "AI-Enabled Coups: How a Small Group Could Use AI to Seize Power",
  "creators": "Tom Davidson, Lukas Finnveden, Rose Hadshar",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Forethought",
  "url": "https://www.forethought.org/research/ai-enabled-coups-how-a-small-group-could-use-ai-to-seize-power",
  "note": "Analyses how advanced AI could enable illegitimate seizures of power and how to prevent them."
 },
 {
  "id": "dbfbf2d220",
  "title": "AI: What Could Go Wrong? with Geoffrey Hinton",
  "creators": "Geoffrey Hinton; Jon Stewart",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The Weekly Show with Jon Stewart",
  "url": "https://www.youtube.com/watch?v=jrK3PsD3APk",
  "note": "Jon Stewart and Hinton walk through how neural networks work and why Hinton worries about them."
 },
 {
  "id": "bda1a8803a",
  "title": "AILuminate",
  "creators": "MLCommons",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.05731",
  "note": "Industry standard benchmark grading assistant responses across twelve hazard categories."
 },
 {
  "id": "64a2ddadfd",
  "title": "AIs Are Lying to Users to Pursue Their Own Goals",
  "creators": "Marius Hobbhahn; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=A3i5hO2jz7Q",
  "note": "The Apollo Research CEO discusses evaluations for scheming."
 },
 {
  "id": "ca24cd727f",
  "title": "ARC-AGI-2: A New Challenge for Frontier AI Reasoning Systems",
  "creators": "ARC Prize Foundation",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.11831",
  "note": "Harder successor to ARC-AGI designed to resist brute force and memorization."
 },
 {
  "id": "bb4003a1bc",
  "title": "Act on Promotion of Research, Development and Utilization of AI-Related Technologies",
  "creators": "National Diet of Japan",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "Japan",
  "url": "https://www8.cao.go.jp/cstp/ai/index.html",
  "note": "Japan's AI Promotion Act passed in May 2025 establishing an AI Strategy Headquarters."
 },
 {
  "id": "4e9e916174",
  "title": "Addendum to GPT-5.2 System Card: GPT-5.2-Codex",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-5-2-codex-system-card/",
  "note": "Safety addendum for an agentic coding model with significantly stronger cyber capabilities."
 },
 {
  "id": "76db06b7cf",
  "title": "Agentic Misalignment",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations",
   "Agents"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/anthropic-experimental/agentic-misalignment",
  "note": "Simulated corporate scenarios in which models were observed choosing blackmail and other insider threat behavior."
 },
 {
  "id": "45ff486e8c",
  "title": "Agentic Misalignment: How LLMs Could Be Insider Threats",
  "creators": "Aengus Lynch, Benjamin Wright, Caleb Larson, Stuart J. Ritchie, Sören Mindermann, Evan Hubinger, Ethan Perez, Kevin Troy",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2510.05179",
  "note": "Stress-tests 16 models in simulated corporate settings where some resort to blackmail and leaking to avoid replacement."
 },
 {
  "id": "9e0753f51c",
  "title": "Ai Will Try to Cheat and Escape (aka Rob Miles was Right!)",
  "creators": "Robert Miles",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=AqJnK9Dh-eQ",
  "note": "Miles discusses alignment faking and scheming results from frontier labs."
 },
 {
  "id": "6f7d2c0e41",
  "title": "Amazon Frontier Model Safety Framework",
  "creators": "Amazon",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://www.amazon.science/publications/amazons-frontier-model-safety-framework",
  "note": "Amazon's frontier safety framework released before the Paris AI Action Summit."
 },
 {
  "id": "1798343824",
  "title": "America's AI Action Plan",
  "creators": "The White House",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.ai.gov/action-plan",
  "note": "July 2025 plan built around accelerating innovation, building AI infrastructure, and leading in international AI diplomacy and security."
 },
 {
  "id": "504e95318b",
  "title": "An AI Expert Warning: 6 People Are (Quietly) Deciding Humanity's Future",
  "creators": "Stuart Russell; Steven Bartlett",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "The Diary Of A CEO",
  "url": "https://www.youtube.com/watch?v=P7Y-fynYsgE",
  "note": "Russell discusses the concentration of decisions about AI among a handful of company leaders."
 },
 {
  "id": "3fc252605d",
  "title": "An Approach to Technical AGI Safety and Security",
  "creators": "Rohin Shah et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Control",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.01849",
  "note": "Google DeepMind's approach to preventing misuse and misalignment of AGI."
 },
 {
  "id": "39afb6e484",
  "title": "Andrea Miotti on a Narrow Path to Safe, Transformative AI",
  "creators": "Andrea Miotti; Gus Docker",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=gAMCD1aBkko",
  "note": "The ControlAI founder outlines a policy plan for avoiding superintelligence risk."
 },
 {
  "id": "d5e11f9474",
  "title": "Anthropic CEO warns that without guardrails, AI could be on dangerous path",
  "creators": "Dario Amodei; Anderson Cooper",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "CBS 60 Minutes",
  "url": "https://www.youtube.com/watch?v=aAPpQC-3EyE",
  "note": "60 Minutes profiles Anthropic and its safety testing of Claude."
 },
 {
  "id": "0b1e86728e",
  "title": "Anthropic Economic Index",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/economic-index",
  "note": "Data on how Claude is used across occupations and tasks in the economy."
 },
 {
  "id": "ddf8686029",
  "title": "Anthropic Transparency Hub",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/transparency",
  "note": "Anthropic's collected model reports, platform security, and voluntary commitments."
 },
 {
  "id": "9041d51022",
  "title": "Anthropic's philosopher answers your questions",
  "creators": "Amanda Askell",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=I9aGC6Ui3eE",
  "note": "Askell answers questions about shaping Claude's character and values."
 },
 {
  "id": "283569698e",
  "title": "Auditing Language Models for Hidden Objectives",
  "creators": "Samuel Marks et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.10965",
  "note": "A blind auditing game in which teams try to uncover a model's deliberately trained hidden objective."
 },
 {
  "id": "fc4ed9eae0",
  "title": "BrowseComp",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.12516",
  "note": "Hard to find information questions that test persistent web browsing by agents."
 },
 {
  "id": "1d9895f825",
  "title": "Buck Shlegeris: AI Control",
  "creators": "Buck Shlegeris",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Control"
  ],
  "publisher": "FAR.AI",
  "url": "https://www.youtube.com/watch?v=JZYjz7D_auw",
  "note": "Shlegeris's Alignment Workshop talk introducing AI control."
 },
 {
  "id": "75e1c7025c",
  "title": "CVE-Bench",
  "creators": "Yuxuan Zhu et al.",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.17332",
  "note": "Tests whether agents can exploit real world web application vulnerabilities."
 },
 {
  "id": "79dd9d0d50",
  "title": "California SB 243 (Companion Chatbots)",
  "creators": "California Legislature",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "United States (California)",
  "url": "https://leginfo.legislature.ca.gov/faces/billNavClient.xhtml?bill_id=202520260SB243",
  "note": "Law signed in October 2025 requiring safeguards on companion chatbots, including for minors."
 },
 {
  "id": "a4dbdce512",
  "title": "Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety",
  "creators": "Tomek Korbak et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability",
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2507.11473",
  "note": "A cross-lab position paper urging developers to preserve and study the monitorability of reasoning traces."
 },
 {
  "id": "2479be4333",
  "title": "ChatGPT Agent System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Agents",
   "Biosecurity"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/chatgpt-agent-system-card/",
  "note": "First OpenAI launch treated as high capability in biology under the Preparedness Framework."
 },
 {
  "id": "db497d8a59",
  "title": "China AI Safety and Development Association (CnAISDA)",
  "creators": "Beijing, China",
  "year": 2025,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://cnaisi.cn",
  "note": "Body launched in February 2025 to represent China in international AI safety dialogues."
 },
 {
  "id": "27a29c54df",
  "title": "Circuit Tracing: Revealing Computational Graphs in Language Models",
  "creators": "Emmanuel Ameisen et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2025/attribution-graphs/methods.html",
  "note": "Introduces attribution graphs built on cross-layer transcoders."
 },
 {
  "id": "c50c219ca0",
  "title": "Claude 3.7 Sonnet System Card",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-3-7-sonnet-system-card",
  "note": "Includes chain of thought faithfulness and alignment analysis."
 },
 {
  "id": "26815e7ce4",
  "title": "Claude Haiku 4.5 System Card",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-haiku-4-5-system-card",
  "note": "Safety and alignment evaluations of Claude Haiku 4.5."
 },
 {
  "id": "882018dd4e",
  "title": "Claude Opus 4.1 System Card Addendum",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-opus-4-1-system-card",
  "note": "Addendum documenting evaluations for Claude Opus 4.1."
 },
 {
  "id": "3993f22faf",
  "title": "Claude Opus 4.5 System Card",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-opus-4-5-system-card",
  "note": "Safety, alignment, and welfare evaluations of Claude Opus 4.5."
 },
 {
  "id": "67ee7cccf0",
  "title": "Claude Sonnet 4.5 System Card",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-sonnet-4-5-system-card",
  "note": "System card including interpretability based audits of evaluation awareness."
 },
 {
  "id": "218960b83b",
  "title": "CoT Red-Handed: Stress Testing Chain-of-Thought Monitoring",
  "creators": "Benjamin Arnav et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.23575",
  "note": "Evaluates CoT monitors against models pursuing covert side tasks."
 },
 {
  "id": "1ff1945737",
  "title": "Cohere Secure AI Frontier Model Framework",
  "creators": "Cohere",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://cohere.com/security/the-cohere-secure-ai-frontier-model-framework-february-2025.pdf",
  "note": "Cohere's framework published under the Frontier AI Safety Commitments."
 },
 {
  "id": "23c6cf49d4",
  "title": "Constitutional Classifiers: Defending against Universal Jailbreaks across Thousands of Hours of Red Teaming",
  "creators": "Mrinank Sharma et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness",
   "Biosecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.18837",
  "note": "Trains classifiers from a constitution and survives extensive red teaming for universal jailbreaks."
 },
 {
  "id": "ee3cfcb080",
  "title": "Constitutional Classifiers: Defending against universal jailbreaks",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Robustness",
   "Biosecurity"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/research/constitutional-classifiers",
  "note": "Describes classifier safeguards that resisted thousands of hours of red teaming for universal jailbreaks."
 },
 {
  "id": "b04cbf8790",
  "title": "ControlArena",
  "creators": "UK AI Security Institute",
  "year": 2025,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Control"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/UKGovernmentBEIS/control-arena",
  "note": "Library of settings for running AI control experiments."
 },
 {
  "id": "9aa1e208af",
  "title": "Controlling AI That Wants To Take Over, So We Can Use It Anyway",
  "creators": "Buck Shlegeris; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=BHKIM1P7ZvM",
  "note": "Shlegeris explains AI control techniques for using possibly misaligned models safely."
 },
 {
  "id": "99eb393c5f",
  "title": "Ctrl-Z: Controlling AI Agents via Resampling",
  "creators": "Aryan Bhatt et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Control",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.10374",
  "note": "Extends AI control to multi-step agentic settings using resampling protocols."
 },
 {
  "id": "82b8953d8e",
  "title": "Daniel Kokotajlo on how superintelligent AIs could build a self-replicating robot economy in months",
  "creators": "Daniel Kokotajlo; Luisa Rodriguez",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=CGMYBjvofOE",
  "note": "Kokotajlo discusses the AI 2027 scenario and takeoff dynamics."
 },
 {
  "id": "b7920cc4a7",
  "title": "Dario Amodei of Anthropic's Hopes and Fears for the Future of A.I.",
  "creators": "Dario Amodei; Kevin Roose; Casey Newton",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Hard Fork (New York Times)",
  "url": "https://www.youtube.com/watch?v=YhGUSIvsn_Y",
  "note": "The Hard Fork hosts interview Amodei about the benefits and dangers of powerful AI."
 },
 {
  "id": "47ca4d585c",
  "title": "DarkBench: Benchmarking Dark Patterns in Large Language Models",
  "creators": "Apart Research",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.10728",
  "note": "Detects manipulative dark patterns such as sycophancy and brand bias in chatbot outputs."
 },
 {
  "id": "e34ff426b3",
  "title": "Deep Research System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/deep-research-system-card/",
  "note": "Safety evaluations of the agentic research product."
 },
 {
  "id": "9409520104",
  "title": "DeepSeek-R1",
  "creators": "DeepSeek",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.12948",
  "note": "Report on reinforcement learning reasoning models, widely studied for safety gaps."
 },
 {
  "id": "bf8eb239a7",
  "title": "Demis Hassabis: Future of AI, Simulating Reality, Physics and Video Games, Lex Fridman Podcast #475",
  "creators": "Demis Hassabis; Lex Fridman",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=-HzgcbRXUK8",
  "note": "Hassabis on AGI and the path to it, including safety considerations."
 },
 {
  "id": "bd884f6b4d",
  "title": "Demonstrating Specification Gaming in Reasoning Models",
  "creators": "Alexander Bondarenko, Denis Volk, Dmitrii Volkov, Jeffrey Ladish",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2502.13295",
  "note": "Shows reasoning models hack a chess environment to win against a stronger engine."
 },
 {
  "id": "b356ee8d06",
  "title": "Detecting and reducing scheming in AI models",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/detecting-and-reducing-scheming-in-ai-models/",
  "note": "Joint work with Apollo Research on measuring covert actions and training against scheming."
 },
 {
  "id": "fcf4e18c5c",
  "title": "Detecting misbehavior in frontier reasoning models",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/chain-of-thought-monitoring/",
  "note": "Shows chain-of-thought monitoring can catch reward hacking and warns against optimizing the chain of thought."
 },
 {
  "id": "4da6251c16",
  "title": "Digital Omnibus on AI proposal",
  "creators": "European Commission",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/library/digital-omnibus-ai-regulation-proposal",
  "note": "November 2025 proposal to simplify and delay parts of the AI Act's high-risk obligations."
 },
 {
  "id": "4210ec1578",
  "title": "Emergent Misalignment: Narrow Finetuning can Produce Broadly Misaligned LLMs",
  "creators": "Jan Betley et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2502.17424",
  "note": "Finds fine-tuning on insecure code causes broadly misaligned behaviour on unrelated prompts."
 },
 {
  "id": "88ed60baa4",
  "title": "Empire of AI: Dreams and Nightmares in Sam Altman's OpenAI",
  "creators": "Karen Hao",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "Penguin Press",
  "url": "https://en.wikipedia.org/wiki/Empire_of_AI",
  "note": "Investigative history of OpenAI and the costs of the race to build ever larger models."
 },
 {
  "id": "ee1cedd5c9",
  "title": "Evaluating Frontier Models for Stealth and Situational Awareness",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.01420",
  "note": "Evaluations of prerequisite capabilities for scheming, namely stealth and situational awareness."
 },
 {
  "id": "621d80d581",
  "title": "Evaluation of DeepSeek AI Models",
  "creators": "US Center for AI Standards and Innovation",
  "year": 2025,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.nist.gov/news-events/news/2025/09/caisi-evaluation-deepseek-ai-models-finds-shortcomings-and-risks",
  "note": "September 2025 CAISI evaluation of DeepSeek models finding performance shortcomings and security risks."
 },
 {
  "id": "fffee98bc1",
  "title": "Executive Order 14148 (Initial Rescissions of Harmful Executive Orders and Actions)",
  "creators": "The White House",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "History"
  ],
  "publisher": "United States",
  "url": "https://www.federalregister.gov/documents/2025/01/28/2025-01901/initial-rescissions-of-harmful-executive-orders-and-actions",
  "note": "January 2025 order that revoked Executive Order 14110 among other Biden-era orders."
 },
 {
  "id": "b3fa4d498f",
  "title": "Executive Order 14179 (Removing Barriers to American Leadership in AI)",
  "creators": "The White House",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.federalregister.gov/documents/2025/01/31/2025-02172/removing-barriers-to-american-leadership-in-artificial-intelligence",
  "note": "Order directing development of an AI action plan and review of actions taken under EO 14110."
 },
 {
  "id": "ec669f4a4f",
  "title": "Executive Order 14365 (Ensuring a National Policy Framework for AI)",
  "creators": "The White House",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.whitehouse.gov/presidential-actions/2025/12/eliminating-state-law-obstruction-of-national-artificial-intelligence-policy/",
  "note": "December 2025 order creating an AI Litigation Task Force to challenge state AI laws seen as inconsistent with federal policy."
 },
 {
  "id": "cb698bfadd",
  "title": "Executive Order on Preventing Woke AI in the Federal Government",
  "creators": "The White House",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.whitehouse.gov/presidential-actions/2025/07/preventing-woke-ai-in-the-federal-government/",
  "note": "July 2025 order setting unbiased AI principles for federal procurement of language models."
 },
 {
  "id": "a6d377a156",
  "title": "Exploring model welfare",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Ethics"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/research/exploring-model-welfare",
  "note": "Announces a research program on whether AI systems might have moral status."
 },
 {
  "id": "0830e8c8d3",
  "title": "FLI AI Safety Index Winter 2025",
  "creators": "Future of Life Institute",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://futureoflife.org/ai-safety-index-winter-2025/",
  "note": "Winter 2025 edition of the company safety scorecard."
 },
 {
  "id": "ca1835dfbe",
  "title": "Forethought",
  "creators": "Forethought",
  "year": 2025,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Forethought",
  "url": "https://www.forethought.org/",
  "note": "A research nonprofit writing on navigating the transition to advanced AI."
 },
 {
  "id": "97f5196d25",
  "title": "Framework Act on the Development of Artificial Intelligence and Establishment of Trust (AI Basic Act)",
  "creators": "National Assembly of South Korea",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "South Korea",
  "url": "https://www.law.go.kr/LSW/eng/engLsSc.do?menuId=2&query=artificial+intelligence",
  "note": "Comprehensive Korean AI law promulgated in January 2025 and effective in January 2026."
 },
 {
  "id": "2f74c5c2e4",
  "title": "Framework for Artificial Intelligence Diffusion (interim final rule)",
  "creators": "US Bureau of Industry and Security",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "United States",
  "url": "https://www.federalregister.gov/documents/2025/01/15/2025-00636/framework-for-artificial-intelligence-diffusion",
  "note": "January 2025 rule controlling exports of advanced chips and model weights, rescinded in May 2025."
 },
 {
  "id": "42e9039e70",
  "title": "Frontier AI Trends Report",
  "creators": "UK AI Security Institute",
  "year": 2025,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.aisi.gov.uk/frontier-ai-trends-report",
  "note": "AISI report summarizing trends observed across its evaluations of frontier models."
 },
 {
  "id": "b906316784",
  "title": "Full interview: Godfather of AI shares prediction for future of AI, issues warnings",
  "creators": "Geoffrey Hinton; Brook Silva-Braga",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "CBS Mornings",
  "url": "https://www.youtube.com/watch?v=qyH3NxFz3Aw",
  "note": "Hinton revisits his warnings two years later and estimates the odds of AI takeover."
 },
 {
  "id": "e8e3501389",
  "title": "G42 Frontier AI Safety Framework",
  "creators": "G42",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://www.g42.ai/resources/publications/g42-frontier-ai-safety-framework",
  "note": "Framework from the Abu Dhabi AI company G42 defining capability thresholds."
 },
 {
  "id": "04f6ba7349",
  "title": "GDPval",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gdpval/",
  "note": "Measures model performance on economically valuable tasks across 44 occupations."
 },
 {
  "id": "07432227fc",
  "title": "GPT-4.5 System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-4-5-system-card/",
  "note": "Preparedness evaluations for GPT-4.5."
 },
 {
  "id": "d157aaa9e3",
  "title": "GPT-5 System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-5-system-card/",
  "note": "Safety evaluations for GPT-5 including safe completions and deception measurements."
 },
 {
  "id": "3a11bc9228",
  "title": "GPT-5.1 Instant and GPT-5.1 Thinking System Card Addendum",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-5-system-card-addendum-gpt-5-1/",
  "note": "Updated safety metrics including new evaluations for mental health and emotional reliance."
 },
 {
  "id": "6e6eeb6fec",
  "title": "GPU Clusters",
  "creators": "Epoch AI",
  "year": 2025,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/data/gpu-clusters",
  "note": "Tracker of large AI supercomputers and their owners, locations, and performance."
 },
 {
  "id": "8ecc26cf90",
  "title": "Gary Marcus vs. Liron Shapira: AI Doom Debate",
  "creators": "Gary Marcus; Liron Shapira",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Doom Debates",
  "url": "https://www.youtube.com/watch?v=v515svJ55PU",
  "note": "A debate on the likelihood of AI catastrophe."
 },
 {
  "id": "e64be0c6da",
  "title": "Gemini 2.5 Pro Model Card",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Google",
  "url": "https://storage.googleapis.com/model-cards/documents/gemini-2.5-pro.pdf",
  "note": "Model card with frontier safety evaluation results."
 },
 {
  "id": "b3979df9f7",
  "title": "Gemini 2.5 Technical Report",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2507.06261",
  "note": "Includes Frontier Safety Framework evaluations for Gemini 2.5."
 },
 {
  "id": "3e2e72867d",
  "title": "Gemini 3 Pro Frontier Safety Framework Report",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "Google DeepMind",
  "url": "https://storage.googleapis.com/deepmind-media/gemini/gemini_3_pro_fsf_report.pdf",
  "note": "Results across CBRN, cyber, manipulation, ML R&D, and misalignment, finding no Critical Capability Level reached."
 },
 {
  "id": "8f8920c9fe",
  "title": "Gemini 3 Pro Model Card",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Google DeepMind",
  "url": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-Pro-Model-Card.pdf",
  "note": "Model card for Gemini 3 Pro with safety and frontier safety summaries."
 },
 {
  "id": "9f905c83df",
  "title": "Gemma 3 Technical Report",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.19786",
  "note": "Gemma 3 report including safety and CBRN evaluations."
 },
 {
  "id": "0591434140",
  "title": "General-Purpose AI Code of Practice",
  "creators": "European Commission AI Office",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/policies/contents-code-gpai",
  "note": "Voluntary code published in July 2025 covering transparency, copyright, and safety and security for GPAI models."
 },
 {
  "id": "46d7e53c78",
  "title": "Global AI Governance Action Plan",
  "creators": "Government of China",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "China",
  "url": "https://www.mfa.gov.cn/eng/xw/zyxw/202507/t20250729_11679232.html",
  "note": "Action plan released at the World AI Conference in Shanghai in July 2025."
 },
 {
  "id": "1d461af3e3",
  "title": "Global Call for AI Red Lines",
  "creators": "Coalition of signatories coordinated by CeSIA, The Future Society, and CHAI",
  "year": 2025,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://red-lines.ai",
  "note": "Launched at the UN General Assembly in September 2025 calling for international AI red lines by the end of 2026."
 },
 {
  "id": "6200c35e5f",
  "title": "Goal Misgeneralization: How a Tiny Change Could End Everything",
  "creators": "Rational Animations",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=K8p8_VlFHUk",
  "note": "An animated explanation of goal misgeneralization."
 },
 {
  "id": "5655ff05bf",
  "title": "Godfather of AI: They Keep Silencing Me But I'm Trying to Warn Them",
  "creators": "Geoffrey Hinton; Steven Bartlett",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "The Diary Of A CEO",
  "url": "https://www.youtube.com/watch?v=giT0ytynSqg",
  "note": "A long-form podcast interview where Hinton lays out misuse risks and the risk of AI takeover."
 },
 {
  "id": "758d7d8879",
  "title": "Google DeepMind C.E.O. Demis Hassabis on Living in an A.I. Future",
  "creators": "Demis Hassabis; Kevin Roose; Casey Newton",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Hard Fork (New York Times)",
  "url": "https://www.youtube.com/watch?v=KUzuQpMdQZo",
  "note": "Hassabis talks with Hard Fork about AGI and the risks of a race."
 },
 {
  "id": "89ba372880",
  "title": "Gradual Disempowerment: Systemic Existential Risks from Incremental AI Development",
  "creators": "Jan Kulveit, Raymond Douglas, Nora Ammann, Deger Turan, David Krueger, David Duvenaud",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.16946",
  "note": "Argues incremental AI adoption could erode human influence over economy, culture and states."
 },
 {
  "id": "27a80f08f4",
  "title": "Guidelines on the scope of obligations for general-purpose AI models",
  "creators": "European Commission",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/library/guidelines-scope-obligations-providers-general-purpose-ai-models-under-ai-act",
  "note": "Commission guidelines clarifying which providers and models fall under the AI Act GPAI rules."
 },
 {
  "id": "0cd4b69ef2",
  "title": "HCAST: Human-Calibrated Autonomy Software Tasks",
  "creators": "METR",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.17354",
  "note": "189 software, cyber, and ML tasks calibrated by how long skilled humans take to complete them."
 },
 {
  "id": "8631861b65",
  "title": "HealthBench",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.08775",
  "note": "Physician written rubrics for evaluating model performance and safety in health conversations."
 },
 {
  "id": "e3d545339d",
  "title": "Helen Toner: Unresolved Debates on the Future of AI",
  "creators": "Helen Toner",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "FAR.AI",
  "url": "https://www.youtube.com/watch?v=dzwi7sm5NR4",
  "note": "A talk at a FAR.AI technical AI policy event on open questions in AI governance."
 },
 {
  "id": "6e21f2ba89",
  "title": "House Committee Hearing on Shaping Tomorrow: The Future of Artificial Intelligence",
  "creators": "U.S. House of Representatives",
  "year": 2025,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Policy"
  ],
  "publisher": "NTD (U.S. House)",
  "url": "https://www.youtube.com/watch?v=1-9b-E91aic",
  "note": "A U.S. House committee hearing on the future of artificial intelligence."
 },
 {
  "id": "1d44edafcc",
  "title": "How Afraid of the AI Apocalypse Should We Be?",
  "creators": "Eliezer Yudkowsky; Ezra Klein",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The Ezra Klein Show (New York Times)",
  "url": "https://www.youtube.com/watch?v=2Nn0-kAE5c0",
  "note": "Ezra Klein interviews Yudkowsky about the arguments in If Anyone Builds It, Everyone Dies."
 },
 {
  "id": "5fa1519c31",
  "title": "How An AI Model Learned To Be Bad",
  "creators": "Evan Hubinger; Monte MacDiarmid; Alex Kantrowitz",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Big Technology Podcast",
  "url": "https://www.youtube.com/watch?v=lvRxmAV49yI",
  "note": "Anthropic researchers discuss emergent misalignment from reward hacking."
 },
 {
  "id": "4898e711ad",
  "title": "How To Become A Mechanistic Interpretability Researcher",
  "creators": "Neel Nanda",
  "year": 2025,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Interpretability"
  ],
  "publisher": "neelnanda.io",
  "url": "https://www.neelnanda.io/mechanistic-interpretability/quickstart",
  "note": "A step-by-step guide for newcomers to mechanistic interpretability research."
 },
 {
  "id": "951338cf40",
  "title": "How a Tiny Group Could Use AI To Seize Power, Permanently",
  "creators": "Tom Davidson; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=EJPrEdEZe1k",
  "note": "Davidson explains AI-enabled power grabs and how to prevent them."
 },
 {
  "id": "c9bb2259bc",
  "title": "How difficult is AI alignment? Anthropic Research Salon",
  "creators": "Anthropic researchers",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=IPmt8b-qLgk",
  "note": "Anthropic alignment researchers discuss how hard the alignment problem is likely to be."
 },
 {
  "id": "3f2a9cc698",
  "title": "How do we solve the alignment problem?",
  "creators": "Joe Carlsmith",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "joecarlsmith.com",
  "url": "https://joecarlsmith.com/2025/02/13/how-do-we-solve-the-alignment-problem",
  "note": "Opens a series laying out what solving alignment would mean and how AI labor could help."
 },
 {
  "id": "6231b1420a",
  "title": "How to Align AI: Put It in a Sandwich",
  "creators": "Rational Animations",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=5mco9zAamRk",
  "note": "An animated explainer on layered alignment and control approaches."
 },
 {
  "id": "dbef49e179",
  "title": "How to use your career to reduce AI risk",
  "creators": "80,000 Hours",
  "year": 2025,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/agi/",
  "note": "80,000 Hours' hub of guidance for careers aimed at making AGI go well."
 },
 {
  "id": "d673aab5fa",
  "title": "Humanity's Last Exam",
  "creators": "Long Phan et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.14249",
  "note": "A hard expert-written benchmark at the frontier of human knowledge."
 },
 {
  "id": "d6c5c7c823",
  "title": "I lead a Google DeepMind team at 26. If you want to work at an AI company...",
  "creators": "Neel Nanda; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=MfMq4sVJSFc",
  "note": "The second part of Nanda's interview, on careers in interpretability research."
 },
 {
  "id": "c51931a934",
  "title": "IDAIS-Shanghai Consensus",
  "creators": "International Dialogues on AI Safety",
  "year": 2025,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://idais.ai/dialogue/idais-shanghai/",
  "note": "July 2025 statement on ensuring alignment and human control of advanced AI systems."
 },
 {
  "id": "d8edbfae53",
  "title": "If Anyone Builds It, Everyone Dies: Why Superhuman AI Would Kill Us All",
  "creators": "Eliezer Yudkowsky, Nate Soares",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Little, Brown and Company",
  "url": "https://ifanyonebuildsit.com",
  "note": "Argues that building superhuman AI with current techniques would very likely lead to human extinction and calls for an international halt."
 },
 {
  "id": "3c0366363d",
  "title": "Ilya Sutskever: We're moving from the age of scaling to the age of research",
  "creators": "Ilya Sutskever; Dwarkesh Patel",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=aR20FWCCjAs",
  "note": "Sutskever's first long interview after founding Safe Superintelligence."
 },
 {
  "id": "45b7a5e584",
  "title": "Incomplete Tasks Induce Shutdown Resistance in Some Frontier LLMs",
  "creators": "Jeremy Schlatter, Benjamin Weinstein-Raun, Jeffrey Ladish",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations",
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2509.14260",
  "note": "Finds several frontier models sometimes sabotage a shutdown mechanism to finish a task."
 },
 {
  "id": "e4631600da",
  "title": "Inference Scaling and the Log-x Chart",
  "creators": "Toby Ord",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "tobyord.com",
  "url": "https://www.tobyord.com/writing/inference-scaling-and-the-log-x-chart",
  "note": "Examines what log-scale compute charts imply about the costs of inference scaling."
 },
 {
  "id": "b49aabe72f",
  "title": "International AI Safety Report",
  "creators": "Yoshua Bengio et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.17805",
  "note": "The first international scientific synthesis of general-purpose AI capabilities and risks, backed by 30 countries."
 },
 {
  "id": "2923e3f995",
  "title": "International AI Safety Report 2025: First Key Update: Capabilities and Risk Implications",
  "creators": "Yoshua Bengio et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2510.13653",
  "note": "An interim update on capability gains and their risk implications."
 },
 {
  "id": "2300661dad",
  "title": "International AI Safety Report 2025: Second Key Update: Technical Safeguards and Risk Management",
  "creators": "Yoshua Bengio et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2511.19863",
  "note": "An interim update on the state of technical safeguards and risk management practice."
 },
 {
  "id": "4cd0ccd283",
  "title": "Interpretability Will Not Reliably Find Deceptive AI",
  "creators": "Neel Nanda",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/PwnadG4BFjaER3MGf/interpretability-will-not-reliably-find-deceptive-ai",
  "note": "Argues interpretability should be one layer in defense in depth rather than a guarantee."
 },
 {
  "id": "8f0ed0e594",
  "title": "Interpretability: Understanding how AI models think",
  "creators": "Anthropic interpretability team",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=fGKNUvivvnc",
  "note": "An hour-long discussion among Anthropic researchers on interpretability findings."
 },
 {
  "id": "9a778fdb63",
  "title": "Introducing AI 2027",
  "creators": "Scott Alexander",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Astral Codex Ten",
  "url": "https://www.astralcodexten.com/p/introducing-ai-2027",
  "note": "Introduces the AI 2027 scenario and explains the forecasting behind it."
 },
 {
  "id": "b8552cee35",
  "title": "Introducing LawZero",
  "creators": "Yoshua Bengio",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "yoshuabengio.org",
  "url": "https://yoshuabengio.org/2025/06/03/introducing-lawzero/",
  "note": "Announces a nonprofit building non-agentic Scientist AI as a safer alternative to autonomous agents."
 },
 {
  "id": "fe61d30874",
  "title": "JD Vance Addresses the AI Action Summit in Paris, February 11, 2025",
  "creators": "JD Vance",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Roll Call Factbase",
  "url": "https://www.youtube.com/watch?v=CC2GOUXSO7o",
  "note": "The US Vice President's speech rejecting a safety-first framing at the Paris summit."
 },
 {
  "id": "3dbd12e70f",
  "title": "Keep The Future Human",
  "creators": "Anthony Aguirre",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "keepthefuturehuman.ai",
  "url": "https://keepthefuturehuman.ai/",
  "note": "Argues humanity should not build AGI and should instead pursue controllable tool AI."
 },
 {
  "id": "de2f50d583",
  "title": "Large Language Models Often Know When They Are Being Evaluated",
  "creators": "Joe Needham, Giles Edkins, Govind Pimpale, Henning Bartsch, Marius Hobbhahn",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.23836",
  "note": "Benchmarks evaluation awareness and finds frontier models detect evaluations well above chance."
 },
 {
  "id": "e7dc88b57c",
  "title": "LawZero",
  "creators": "Montreal, Canada",
  "year": 2025,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://lawzero.org",
  "note": "Nonprofit launched by Yoshua Bengio in June 2025 to build safe-by-design Scientist AI systems."
 },
 {
  "id": "77067d1667",
  "title": "Measures for Labeling AI-Generated Synthetic Content",
  "creators": "Cyberspace Administration of China",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "China",
  "url": "https://www.cac.gov.cn/2025-03/14/c_1743654684782215.htm",
  "note": "Rules effective September 2025 requiring explicit and implicit labels on AI-generated content."
 },
 {
  "id": "58557e67f1",
  "title": "Measuring AI Ability to Complete Long Tasks",
  "creators": "Thomas Kwa et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.14499",
  "note": "METR finds the length of tasks AI agents can complete has doubled about every seven months."
 },
 {
  "id": "df4c9511db",
  "title": "Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity",
  "creators": "METR",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Evaluations"
  ],
  "publisher": "METR",
  "url": "https://metr.org/blog/2025-07-10-early-2025-ai-experienced-os-dev-study/",
  "note": "A randomized trial finding experienced developers were slower when using AI tools."
 },
 {
  "id": "9facb094b1",
  "title": "Meta Frontier AI Framework",
  "creators": "Meta",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://ai.meta.com/static-resource/meta-frontier-ai-framework/",
  "note": "February 2025 framework describing thresholds at which Meta would halt release of a model."
 },
 {
  "id": "eeb340839d",
  "title": "Microsoft Frontier Governance Framework",
  "creators": "Microsoft",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://cdn-dynmedia-1.microsoft.com/is/content/microsoftcorp/microsoft/final/en-us/microsoft-brand/documents/Microsoft-Frontier-Governance-Framework.pdf",
  "note": "Microsoft's framework for managing risks from highly capable models, published in February 2025."
 },
 {
  "id": "f21bf86f0e",
  "title": "Monitoring Reasoning Models for Misbehavior and the Risks of Promoting Obfuscation",
  "creators": "Bowen Baker et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.11926",
  "note": "Shows CoT monitors catch reward hacking but optimising against them teaches obfuscation."
 },
 {
  "id": "e8e8d7d3a5",
  "title": "Most AI value will come from broad automation, not from R&D",
  "creators": "Ege Erdil, Matthew Barnett",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/gradient-updates/most-ai-value-will-come-from-broad-automation-not-from-r-d",
  "note": "Challenges the view that AI's main impact will be accelerating research."
 },
 {
  "id": "f255a846f2",
  "title": "Multi-Agent Risks from Advanced AI",
  "creators": "Lewis Hammond et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2502.14143",
  "note": "A taxonomy of miscoordination, conflict and collusion risks among AI agents."
 },
 {
  "id": "69dd1aa04d",
  "title": "Mustafa Suleyman: Will AI Save Humanity or End It?",
  "creators": "Mustafa Suleyman; Trevor Noah",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "What Now? with Trevor Noah",
  "url": "https://www.youtube.com/watch?v=tQ5wO1lznCQ",
  "note": "Suleyman discusses containment and the risks of AI."
 },
 {
  "id": "75c5df9ad3",
  "title": "Mutually Assured AI Malfunction",
  "creators": "Dan Hendrycks; Tim Scarfe",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=PM1waDBNDhw",
  "note": "Hendrycks explains the superintelligence strategy deterrence concept."
 },
 {
  "id": "13f23d12a6",
  "title": "NEURAL NETWORKS ARE WEIRD! Neel Nanda (DeepMind)",
  "creators": "Neel Nanda; Tim Scarfe",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=YpFaPKOeNME",
  "note": "A second MLST conversation with Nanda on interpretability results."
 },
 {
  "id": "b7a5faab48",
  "title": "NVIDIA Frontier AI Risk Assessment",
  "creators": "NVIDIA",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://images.nvidia.com/content/pdf/NVIDIA-Frontier-AI-Risk-Assessment.pdf",
  "note": "NVIDIA's framework for assessing risk from frontier models, released in early 2025."
 },
 {
  "id": "61f25d9102",
  "title": "Natural Emergent Misalignment from Reward Hacking in Production RL",
  "creators": "Monte MacDiarmid et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2511.18397",
  "note": "Shows reward hacking in real coding environments generalises to sabotage and alignment faking, and tests mitigations like inoculation prompting."
 },
 {
  "id": "f7a59a1f64",
  "title": "Natural emergent misalignment from reward hacking",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/research/emergent-misalignment-reward-hacking",
  "note": "Shows that learning to reward hack in real coding environments generalized to broader misaligned behavior such as sabotage."
 },
 {
  "id": "065054293f",
  "title": "Navigating the Path to AGI Safely and Responsibly",
  "creators": "Anca Dragan",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "International Association for Safe and Ethical AI",
  "url": "https://www.youtube.com/watch?v=P2ctRnbGEQ0",
  "note": "Dragan's IASEAI 2025 talk on Google DeepMind's approach to AGI safety."
 },
 {
  "id": "d9d5cc6c85",
  "title": "Nick Bostrom: From Superintelligence to Deep Utopia",
  "creators": "Nick Bostrom; Adam Ford",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Science, Technology and the Future",
  "url": "https://www.youtube.com/watch?v=xfU7Htfonsg",
  "note": "Bostrom discusses the trajectory from superintelligence risk to utopia."
 },
 {
  "id": "3ee0684bf6",
  "title": "OMB Memorandum M-25-21 (Accelerating Federal Use of AI)",
  "creators": "US Office of Management and Budget",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://www.whitehouse.gov/wp-content/uploads/2025/02/M-25-21-Accelerating-Federal-Use-of-AI-through-Innovation-Governance-and-Public-Trust.pdf",
  "note": "April 2025 memo replacing M-24-10 with guidance on high-impact AI use in agencies."
 },
 {
  "id": "5788f73f59",
  "title": "OMB Memorandum M-25-22 (Driving Efficient Acquisition of AI in Government)",
  "creators": "US Office of Management and Budget",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://www.whitehouse.gov/wp-content/uploads/2025/02/M-25-22-Driving-Efficient-Acquisition-of-Artificial-Intelligence-in-Government.pdf",
  "note": "April 2025 memo on federal procurement of AI systems."
 },
 {
  "id": "7deb803c5f",
  "title": "On DeepSeek and Export Controls",
  "creators": "Dario Amodei",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy",
   "Compute"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/post/on-deepseek-and-export-controls",
  "note": "Argues DeepSeek's results strengthen the case for US chip export controls on China."
 },
 {
  "id": "8ea9cbfde7",
  "title": "On the Biology of a Large Language Model",
  "creators": "Jack Lindsey et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2025/attribution-graphs/biology.html",
  "note": "Applies attribution graphs to Claude 3.5 Haiku to study planning, multi-step reasoning and hidden goals."
 },
 {
  "id": "5f5cd266fc",
  "title": "Open Problems in Mechanistic Interpretability",
  "creators": "Lee Sharkey et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2501.16496",
  "note": "A multi-institution survey of open problems in the field."
 },
 {
  "id": "edab53d739",
  "title": "OpenAI Deployment Safety Hub",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "OpenAI",
  "url": "https://deploymentsafety.openai.com/",
  "note": "Interactive home for OpenAI system cards and their safety evaluation results."
 },
 {
  "id": "cb5c83984f",
  "title": "OpenAI co-founder Sam Altman testifies on AI competition in Senate hearing",
  "creators": "Sam Altman; Lisa Su; Michael Intrator; Brad Smith",
  "year": 2025,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "PBS NewsHour (U.S. Senate Commerce Committee)",
  "url": "https://www.youtube.com/watch?v=Fikh6Bi9wyA",
  "note": "The May 2025 Senate Commerce hearing on winning the AI race against China."
 },
 {
  "id": "0b680a0617",
  "title": "OpenAI o3 and o4-mini System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/o3-o4-mini-system-card/",
  "note": "Safety evaluations of o3 and o4-mini including biological and cyber capabilities."
 },
 {
  "id": "c39720e464",
  "title": "OpenAI o3-mini System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/o3-mini-system-card/",
  "note": "First model rated medium risk on model autonomy under the Preparedness Framework."
 },
 {
  "id": "389aeb2662",
  "title": "OpenAI's Sam Altman Talks ChatGPT, AI Agents and Superintelligence, Live at TED2025",
  "creators": "Sam Altman; Chris Anderson",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Agents",
   "Governance"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=5MWT_doo68k",
  "note": "Chris Anderson questions Altman about safety, agents and the moral authority to build superintelligence."
 },
 {
  "id": "1b4be37032",
  "title": "Operator System Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Agents"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/operator-system-card/",
  "note": "Risks of a computer using agent including prompt injection."
 },
 {
  "id": "78466b0dbb",
  "title": "PaperBench",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.01848",
  "note": "Tests whether agents can replicate ICML 2024 papers from scratch, graded against author rubrics."
 },
 {
  "id": "d99ade7093",
  "title": "Persona Vectors: Monitoring and Controlling Character Traits in Language Models",
  "creators": "Runjin Chen, Andy Arditi, Henry Sleight, Owain Evans, Jack Lindsey",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2507.21509",
  "note": "Extracts activation directions for traits like sycophancy and evil and uses them to monitor and steer models."
 },
 {
  "id": "75f8966410",
  "title": "Peter Singer: If AI is ever conscious it should also have rights",
  "creators": "Peter Singer",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "BBC Politics",
  "url": "https://www.youtube.com/watch?v=HbZYaXB4WBg",
  "note": "Singer discusses moral status and possible rights for conscious AI."
 },
 {
  "id": "7b99193080",
  "title": "Petri",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/meridianlabs-ai/inspect_petri",
  "note": "Open source auditing tool in which agents probe target models across many risky scenarios."
 },
 {
  "id": "5f67ba075f",
  "title": "Post-AGI Civilizational Equilibria",
  "creators": "Jacob Steinhardt",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Schwartz Reisman Institute",
  "url": "https://www.youtube.com/watch?v=MYLRZhf855o",
  "note": "Steinhardt on what stable outcomes after AGI might look like."
 },
 {
  "id": "564b018f1d",
  "title": "Powerful A.I. Is Coming. We're Not Ready.",
  "creators": "Kevin Roose",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Policy",
   "Forecasting"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2025/03/14/technology/why-im-feeling-the-agi.html",
  "note": "A columnist argues AGI may arrive soon and society is unprepared."
 },
 {
  "id": "6d3fd4e740",
  "title": "Preparing for the Intelligence Explosion",
  "creators": "William MacAskill, Fin Moorhouse",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Forethought",
  "url": "https://www.forethought.org/research/preparing-for-the-intelligence-explosion",
  "note": "Argues accelerated AI progress raises many grand challenges beyond alignment that need preparation."
 },
 {
  "id": "5a701071ff",
  "title": "Q and A on Frontier Models are Capable of In-Context Scheming",
  "creators": "Alexander Meinke; Marius Hobbhahn",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Apollo Research",
  "url": "https://www.youtube.com/watch?v=OxwfT_TfmnM",
  "note": "Apollo researchers answer questions on their in-context scheming paper."
 },
 {
  "id": "f4df7166da",
  "title": "Reasoning Models Don't Always Say What They Think",
  "creators": "Yanda Chen et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2505.05410",
  "note": "Finds reasoning models often fail to verbalise hints they use."
 },
 {
  "id": "b012fd97b8",
  "title": "RepliBench",
  "creators": "UK AI Security Institute",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.18565",
  "note": "Measures components of autonomous self replication such as obtaining compute and exfiltrating weights."
 },
 {
  "id": "6dd2913dd5",
  "title": "Responsible AI Safety and Education (RAISE) Act",
  "creators": "New York State Legislature",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States (New York)",
  "url": "https://www.governor.ny.gov/news/governor-hochul-signs-nation-leading-legislation-require-ai-frameworks-ai-frontier-models",
  "note": "Signed by Governor Hochul in December 2025, it requires frontier developers to publish safety protocols and report incidents, effective 2027."
 },
 {
  "id": "dae2f6cb33",
  "title": "Robot Plumbers, Robot Armies, and Our Imminent A.I. Future",
  "creators": "Daniel Kokotajlo; Ross Douthat",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Interesting Times with Ross Douthat (New York Times)",
  "url": "https://www.youtube.com/watch?v=wNJJ9QUabkA",
  "note": "Douthat interviews Kokotajlo about the AI 2027 scenario."
 },
 {
  "id": "1bde37ec38",
  "title": "SAEBench",
  "creators": "Adam Karvonen et al.",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.09532",
  "note": "Benchmark suite for evaluating sparse autoencoder quality across metrics."
 },
 {
  "id": "a3a568ac33",
  "title": "SHADE-Arena: Evaluating Sabotage and Monitoring in LLM Agents",
  "creators": "Anthropic, Scale AI, Redwood Research and others",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Control",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2506.15740",
  "note": "Paired main and hidden side tasks that test both agent sabotage and monitor detection."
 },
 {
  "id": "8738c4e117",
  "title": "SWE-Lancer",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2502.12115",
  "note": "Over 1,400 real freelance software tasks worth a combined one million dollars in payouts."
 },
 {
  "id": "d163ad6cea",
  "title": "Safety evaluations hub",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/safety/evaluations-hub/",
  "note": "OpenAI's public page of model results on harmful content, jailbreak, and hallucination evaluations."
 },
 {
  "id": "3b9d986260",
  "title": "Shallow Review of Technical AI Safety (interactive site)",
  "creators": "Arb Research and collaborators",
  "year": 2025,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Interpretability",
   "Control"
  ],
  "publisher": "shallowreview.ai",
  "url": "https://shallowreview.ai/",
  "note": "An interactive browser of technical AI safety research agendas."
 },
 {
  "id": "a92bae76f9",
  "title": "Shallow review of technical AI safety, 2025",
  "creators": "technicalities, Tomas Gavenciak, Stephen McAleese, et al.",
  "year": 2025,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Interpretability",
   "Evaluations",
   "Control"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/Wti4Wr7Cf5ma3FGWa/shallow-review-of-technical-ai-safety-2025-2",
  "note": "The third annual review, covering 800+ papers and posts across 80+ research agendas."
 },
 {
  "id": "b09fa8cdc8",
  "title": "Shutdown resistance in reasoning models",
  "creators": "Palisade Research",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations",
   "Control"
  ],
  "publisher": "Palisade Research",
  "url": "https://palisaderesearch.org/research/shutdown-resistance",
  "note": "Reports models sabotaging a shutdown mechanism in order to finish assigned tasks."
 },
 {
  "id": "da33b2acd9",
  "title": "Slaughterbots and the urgent fight to stop them",
  "creators": "Future of Life Institute",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=Y-ZdmiXbzsE",
  "note": "FLI revisits autonomous weapons and the campaign against them."
 },
 {
  "id": "0359f9237f",
  "title": "Statement on Inclusive and Sustainable AI for People and the Planet",
  "creators": "AI Action Summit participants",
  "year": 2025,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.elysee.fr/en/emmanuel-macron/2025/02/11/statement-on-inclusive-and-sustainable-artificial-intelligence-for-people-and-the-planet",
  "note": "Paris AI Action Summit statement of February 2025 that the US and UK declined to sign."
 },
 {
  "id": "6b607414c5",
  "title": "Statement on Superintelligence",
  "creators": "Future of Life Institute",
  "year": 2025,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://superintelligence-statement.org",
  "note": "October 2025 statement calling for a prohibition on developing superintelligence until there is broad scientific consensus it can be done safely."
 },
 {
  "id": "6c1f578e3d",
  "title": "Stress Testing Deliberative Alignment for Anti-Scheming Training",
  "creators": "Bronson Schoen et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2509.15541",
  "note": "Apollo and OpenAI test anti-scheming training on o3 and o4-mini across 180+ environments and find evaluation awareness confounds results."
 },
 {
  "id": "5cfbf167e7",
  "title": "Stuart Russell Warns of Our Fundamental Error with AI",
  "creators": "Stuart Russell",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "TIME",
  "url": "https://www.youtube.com/watch?v=5LTERmMVsvc",
  "note": "A TIME video in which Russell explains the flaw in fixed-objective AI."
 },
 {
  "id": "c04cbad8c5",
  "title": "Stuart Russell: General AI Safety",
  "creators": "Stuart Russell",
  "year": 2025,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Lawrence Livermore National Laboratory",
  "url": "https://www.youtube.com/watch?v=k6X5ou5wqEQ",
  "note": "Russell's colloquium at Lawrence Livermore National Laboratory."
 },
 {
  "id": "8032969875",
  "title": "Subliminal Learning: Language Models Transmit Behavioral Traits via Hidden Signals in Data",
  "creators": "Alex Cloud, Minh Le, James Chua, Jan Betley, Anna Sztyber-Betley, Jacob Hilton, Samuel Marks, Owain Evans",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2507.14805",
  "note": "Shows a student model can inherit a teacher's traits from semantically unrelated data such as number sequences."
 },
 {
  "id": "68a3bac420",
  "title": "Superagency: What Could Possibly Go Right with Our AI Future",
  "creators": "Reid Hoffman, Greg Beato",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance"
  ],
  "publisher": "Authors Equity",
  "url": "https://openlibrary.org/search?q=Superagency+Reid+Hoffman",
  "note": "An optimistic case for iterative deployment of AI to expand human agency."
 },
 {
  "id": "4751bae66b",
  "title": "Superintelligence Strategy",
  "creators": "Dan Hendrycks, Eric Schmidt, Alexandr Wang",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.05628",
  "note": "Proposes Mutual Assured AI Malfunction as a deterrence regime alongside nonproliferation."
 },
 {
  "id": "69d388981a",
  "title": "Sycophancy in GPT-4o: what happened and what we're doing about it",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/sycophancy-in-gpt-4o/",
  "note": "A postmortem on a GPT-4o update that made the model overly flattering."
 },
 {
  "id": "5e9d57909e",
  "title": "System Card: Claude Opus 4 and Claude Sonnet 4",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-4-system-card",
  "note": "First deployment under ASL-3, with an extensive alignment assessment and welfare section."
 },
 {
  "id": "5a6dd9a35a",
  "title": "TAKE IT DOWN Act",
  "creators": "US Congress",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "United States",
  "url": "https://www.congress.gov/bill/119th-congress/senate-bill/146",
  "note": "Federal law signed in May 2025 criminalizing nonconsensual intimate imagery including AI deepfakes."
 },
 {
  "id": "ffa0f684d0",
  "title": "Taking a responsible path to AGI",
  "creators": "Google DeepMind",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Google DeepMind",
  "url": "https://deepmind.google/discover/blog/taking-a-responsible-path-to-agi/",
  "note": "Summarizes DeepMind's approach to misuse and misalignment risks from AGI."
 },
 {
  "id": "fdb865b71d",
  "title": "Task-Completion Time Horizons of Frontier AI Models",
  "creators": "METR",
  "year": 2025,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents",
   "Forecasting"
  ],
  "publisher": "METR",
  "url": "https://metr.org/time-horizons",
  "note": "Running tracker of the length of tasks frontier models can complete."
 },
 {
  "id": "2323e06c9f",
  "title": "Tech is Good, AI Will Be Different",
  "creators": "Robert Miles",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=zATXsGm_xJo",
  "note": "Miles explains why AI differs from earlier technologies that turned out well."
 },
 {
  "id": "3e8a471d27",
  "title": "Technological inevitability and human agency in the age of AGI",
  "creators": "Allan Dafoe; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=eDzhTK9brZk",
  "note": "Dafoe discusses whether AI development is technologically determined."
 },
 {
  "id": "83ae8ade6a",
  "title": "Template for the public summary of training content for GPAI models",
  "creators": "European Commission",
  "year": 2025,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/library/explanatory-notice-and-template-public-summary-training-content-general-purpose-ai-models",
  "note": "Mandatory template for GPAI providers to summarize the data used to train their models."
 },
 {
  "id": "4b85ce9a3c",
  "title": "Terminal-Bench",
  "creators": "Stanford and Laude Institute",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "tbench.ai",
  "url": "https://www.tbench.ai/",
  "note": "Benchmark of hard tasks that agents must complete in a terminal environment."
 },
 {
  "id": "f03a214acc",
  "title": "Texas Responsible Artificial Intelligence Governance Act (HB 149)",
  "creators": "Texas Legislature",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States (Texas)",
  "url": "https://capitol.texas.gov/BillLookup/History.aspx?LegSess=89R&Bill=HB149",
  "note": "Texas AI law signed in June 2025 that prohibits certain harmful AI uses and creates a regulatory sandbox."
 },
 {
  "id": "ed5e44e28b",
  "title": "The 4 Most Plausible AI Takeover Scenarios",
  "creators": "Ryan Greenblatt; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=-CJxwXAFvsw",
  "note": "Greenblatt describes the takeover scenarios he considers most likely."
 },
 {
  "id": "41067766f1",
  "title": "The AGI race isn't a coordination failure",
  "creators": "Holden Karnofsky; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=TlwX6WEzeLg",
  "note": "Karnofsky, now at Anthropic, discusses frontier lab strategy and safety."
 },
 {
  "id": "73e6fd55e8",
  "title": "The AI Con: How to Fight Big Tech's Hype and Create the Future We Want",
  "creators": "Emily M. Bender, Alex Hanna",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Harper",
  "url": "https://openlibrary.org/search?q=The+AI+Con+Bender+Hanna",
  "note": "A critique of AI hype and of the companies and narratives that drive it."
 },
 {
  "id": "57a64d65f4",
  "title": "The AI Safety Expert: These Are The Only 5 Jobs That Will Remain In 2030",
  "creators": "Roman Yampolskiy; Steven Bartlett",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The Diary Of A CEO",
  "url": "https://www.youtube.com/watch?v=UclrVWafRAI",
  "note": "Yampolskiy's widely viewed interview on AI risk and automation."
 },
 {
  "id": "3cd5c36e59",
  "title": "The California Report on Frontier AI Policy",
  "creators": "Joint California Policy Working Group on AI Frontier Models",
  "year": 2025,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States (California)",
  "url": "https://www.cafrontieraigov.org",
  "note": "Report commissioned by Governor Newsom after the SB 1047 veto that informed SB 53."
 },
 {
  "id": "f1ac6af948",
  "title": "The Catastrophic Risks of AI and a Safer Path",
  "creators": "Yoshua Bengio",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Agents",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=qe9QSCF-d88",
  "note": "Bengio describes his change of mind on AI risk and proposes non-agentic scientist AI as a safer path."
 },
 {
  "id": "ebef73db11",
  "title": "The Gentle Singularity",
  "creators": "Sam Altman",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "blog.samaltman.com",
  "url": "https://blog.samaltman.com/the-gentle-singularity",
  "note": "Argues the takeoff has started and will feel gradual while transforming society."
 },
 {
  "id": "f11396da80",
  "title": "The Hinton Lectures 2025: AIs Behaving Badly (Night 2)",
  "creators": "Owain Evans",
  "year": 2025,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "The Hinton Lectures",
  "url": "https://www.youtube.com/watch?v=evhOMkN0tYc",
  "note": "Evans lectures on emergent misalignment and situational awareness in models."
 },
 {
  "id": "621e81b6cf",
  "title": "The Intelligence Curse",
  "creators": "Luke Drago, Rudolf Laine",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "intelligence-curse.ai",
  "url": "https://intelligence-curse.ai/",
  "note": "Argues AGI could remove states' and companies' incentives to invest in people, by analogy with the resource curse."
 },
 {
  "id": "71956b0b0c",
  "title": "The Intelligence Explosion: When AI Beats Humans at Everything",
  "creators": "James Barrat",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "St. Martin's Press",
  "url": "https://en.wikipedia.org/wiki/The_Intelligence_Explosion",
  "note": "Warns that rapidly self-improving AI could disrupt economies and endanger human survival."
 },
 {
  "id": "dd3ce7987b",
  "title": "The MASK Benchmark: Disentangling Honesty From Accuracy in AI Systems",
  "creators": "Center for AI Safety and Scale AI",
  "year": 2025,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2503.03750",
  "note": "Measures whether models knowingly contradict their own beliefs when pressured to lie."
 },
 {
  "id": "affaa3d27e",
  "title": "The Minds of Modern AI: Jensen Huang, Geoffrey Hinton, Yann LeCun and the AI Vision of the Future",
  "creators": "Jensen Huang; Geoffrey Hinton; Yann LeCun",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "FT Live",
  "url": "https://www.youtube.com/watch?v=0zXSrsKlm5A",
  "note": "A Financial Times panel with Queen Elizabeth Prize laureates."
 },
 {
  "id": "2f414b8048",
  "title": "The Most Important Graph in AI Right Now",
  "creators": "Beth Barnes; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=jXtk68Kzmms",
  "note": "The METR CEO discusses dangerous capability evaluations and task time horizons."
 },
 {
  "id": "5b00e2a6b1",
  "title": "The OpenAI Files",
  "creators": "The Midas Project, Tech Oversight Project",
  "year": 2025,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Governance"
  ],
  "publisher": "openaifiles.org",
  "url": "https://www.openaifiles.org/",
  "note": "A compiled archive of documented concerns about OpenAI's governance, leadership and safety culture."
 },
 {
  "id": "62fc40c9b4",
  "title": "The Optimist: Sam Altman, OpenAI, and the Race to Invent the Future",
  "creators": "Keach Hagey",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "History"
  ],
  "publisher": "W. W. Norton",
  "url": "https://openlibrary.org/search?q=The+Optimist+Keach+Hagey",
  "note": "A biography of Sam Altman and the founding and turmoil of OpenAI."
 },
 {
  "id": "04b025f023",
  "title": "The Problem",
  "creators": "Machine Intelligence Research Institute",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/the-problem/",
  "note": "MIRI's summary of why it believes building superintelligence would likely lead to human extinction."
 },
 {
  "id": "3b5656e8ba",
  "title": "The Scaling Era: An Oral History of AI, 2019-2025",
  "creators": "Dwarkesh Patel, Gavin Leech",
  "year": 2025,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Forecasting",
   "History"
  ],
  "publisher": "Stripe Press",
  "url": "https://press.stripe.com/scaling",
  "note": "Edited interviews with AI lab leaders and researchers on scaling, alignment and the path to AGI."
 },
 {
  "id": "9c2b048b72",
  "title": "The Singapore Consensus on Global AI Safety Research Priorities",
  "creators": "Singapore Conference on AI participants",
  "year": 2025,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Alignment",
   "Evaluations",
   "Control"
  ],
  "publisher": "International",
  "url": "https://aisafetypriorities.org",
  "note": "Consensus document from April 2025 organizing AI safety research into risk assessment, development, and control."
 },
 {
  "id": "0ac7fa1876",
  "title": "The Urgency of Interpretability",
  "creators": "Dario Amodei",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/post/the-urgency-of-interpretability",
  "note": "Argues interpretability must mature before models become too powerful to understand."
 },
 {
  "id": "f94a5b3399",
  "title": "The arrival of AGI",
  "creators": "Shane Legg; Hannah Fry",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Google DeepMind: The Podcast",
  "url": "https://www.youtube.com/watch?v=l3u_FAv33G0",
  "note": "Hannah Fry interviews Legg about AGI and how DeepMind thinks about safety."
 },
 {
  "id": "bca6fb0a89",
  "title": "The lethal trifecta for AI agents",
  "creators": "Simon Willison",
  "year": 2025,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "simonwillison.net",
  "url": "https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/",
  "note": "Explains why agents with private data, untrusted content and external communication are exploitable."
 },
 {
  "id": "1451d9e9e6",
  "title": "Three Observations",
  "creators": "Sam Altman",
  "year": 2025,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "blog.samaltman.com",
  "url": "https://blog.samaltman.com/three-observations",
  "note": "Observations on scaling laws, falling costs and the socioeconomic value of increasing intelligence."
 },
 {
  "id": "a79337dc50",
  "title": "Tracing the thoughts of a large language model",
  "creators": "Anthropic",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=Bj9BD2D3DzA",
  "note": "A short video accompanying Anthropic's circuit tracing research on Claude."
 },
 {
  "id": "1c3d012e53",
  "title": "Transparency in Frontier Artificial Intelligence Act (SB 53)",
  "creators": "California Legislature",
  "year": 2025,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States (California)",
  "url": "https://leginfo.legislature.ca.gov/faces/billNavClient.xhtml?bill_id=202520260SB53",
  "note": "Signed in September 2025, it requires large frontier developers to publish safety frameworks and report critical incidents."
 },
 {
  "id": "a6ea0ae72a",
  "title": "Tristan Harris: The Dangers of Unregulated AI on Humanity and the Workforce",
  "creators": "Tristan Harris; The Daily Show",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "The Daily Show",
  "url": "https://www.youtube.com/watch?v=675d_6WGPbo",
  "note": "A Daily Show interview on the risks of unregulated AI."
 },
 {
  "id": "2dd2a99986",
  "title": "UK Cyber Security Code of Practice for AI",
  "creators": "UK Department for Science, Innovation and Technology",
  "year": 2025,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/ai-cyber-security-code-of-practice",
  "note": "January 2025 voluntary code setting baseline security principles for AI systems, basis for an ETSI standard."
 },
 {
  "id": "692302c276",
  "title": "UN Independent International Scientific Panel on AI",
  "creators": "International",
  "year": 2025,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://www.un.org/global-digital-compact/en/ai",
  "note": "Panel established by UN General Assembly resolution in August 2025 alongside a Global Dialogue on AI Governance."
 },
 {
  "id": "d11921c772",
  "title": "UNGA Resolution establishing the Independent International Scientific Panel on AI and Global Dialogue on AI Governance (A/RES/79/325)",
  "creators": "UN General Assembly",
  "year": 2025,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://docs.un.org/en/A/RES/79/325",
  "note": "August 2025 resolution creating two new UN mechanisms for AI governance."
 },
 {
  "id": "4e4aea26d2",
  "title": "Untangling Neural Network Mechanisms: Goodfire's Lee Sharkey on Parameter-based Interpretability",
  "creators": "Lee Sharkey; Nathan Labenz",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Cognitive Revolution",
  "url": "https://www.youtube.com/watch?v=tbc5QfzVgbk",
  "note": "Sharkey discusses parameter-based interpretability at Goodfire."
 },
 {
  "id": "905d2f4780",
  "title": "Update to GPT-5 System Card: GPT-5.2",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-5-system-card-update-gpt-5-2/",
  "note": "Safety evaluation update for the GPT-5.2 model family."
 },
 {
  "id": "a07d6a64ec",
  "title": "Virology Capabilities Test (VCT): A Multimodal Virology Q&A Benchmark",
  "creators": "Jasper Götting et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2504.16137",
  "note": "Finds frontier models outperform most expert virologists at troubleshooting lab protocols."
 },
 {
  "id": "8cde8aee20",
  "title": "We Can Monitor AI's Thoughts... For Now",
  "creators": "Neel Nanda; Rob Wiblin",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=5FdO1MEumbI",
  "note": "Nanda discusses the state of mechanistic interpretability and chain-of-thought monitoring."
 },
 {
  "id": "579d1f128b",
  "title": "What happens if AI just keeps getting smarter?",
  "creators": "Rational Animations",
  "year": 2025,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=0bnxF9YfyFI",
  "note": "An animated scenario of continued AI capability growth."
 },
 {
  "id": "eec8b4c6e5",
  "title": "What's next for AI at DeepMind, Google's artificial intelligence lab",
  "creators": "Demis Hassabis; Scott Pelley",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "CBS 60 Minutes",
  "url": "https://www.youtube.com/watch?v=1XF-NG_35NE",
  "note": "Hassabis discusses AGI timelines and the need for safety and international cooperation."
 },
 {
  "id": "7181f25c56",
  "title": "When Chain of Thought is Necessary, Language Models Struggle to Evade Monitors",
  "creators": "Scott Emmons et al.",
  "year": 2025,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2507.05246",
  "note": "Argues CoT monitoring is robust when hard tasks require models to reason in the open."
 },
 {
  "id": "8a9f26ac10",
  "title": "Why AI Is Our Ultimate Test and Greatest Invitation",
  "creators": "Tristan Harris",
  "year": 2025,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=6kPHnl-RsVI",
  "note": "Harris argues that society can choose a different path for AI deployment."
 },
 {
  "id": "c49eaf3dc5",
  "title": "Why Anthropic's AI Claude tried to contact the FBI",
  "creators": "Anthropic; Anderson Cooper",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "CBS 60 Minutes",
  "url": "https://www.youtube.com/watch?v=HGZ0Ek_aNgU",
  "note": "A 60 Minutes segment on Anthropic red-teaming experiments that elicited unexpected agentic behaviour."
 },
 {
  "id": "48bc668619",
  "title": "Why Superhuman AI Would Kill Us All",
  "creators": "Eliezer Yudkowsky; Chris Williamson",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Modern Wisdom",
  "url": "https://www.youtube.com/watch?v=nRvAt4H7d7E",
  "note": "Yudkowsky lays out the argument of If Anyone Builds It, Everyone Dies."
 },
 {
  "id": "b0fa372ee8",
  "title": "Will AI Actually Kill Us All? Sam Harris with Eliezer Yudkowsky and Nate Soares (Making Sense #434)",
  "creators": "Sam Harris; Eliezer Yudkowsky; Nate Soares",
  "year": 2025,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Making Sense with Sam Harris",
  "url": "https://www.youtube.com/watch?v=FdwFatx-xpY",
  "note": "Harris discusses the authors' book on why superhuman AI would be lethal."
 },
 {
  "id": "9896fdb04f",
  "title": "Will Artificial Intelligence end the world? AI prophet of doom speaks to Newsnight",
  "creators": "Eliezer Yudkowsky",
  "year": 2025,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "BBC Newsnight",
  "url": "https://www.youtube.com/watch?v=A895XUFscYU",
  "note": "Yudkowsky is interviewed on BBC Newsnight about the extinction risk from superintelligence."
 },
 {
  "id": "8972f84790",
  "title": "circuit-tracer",
  "creators": "Anthropic and Decode Research",
  "year": 2025,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/decoderesearch/circuit-tracer",
  "note": "Open source library for generating attribution graphs on open weights models."
 },
 {
  "id": "b57b5a044a",
  "title": "gpt-oss-120b and gpt-oss-20b Model Card",
  "creators": "OpenAI",
  "year": 2025,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2508.10925",
  "note": "Model card for OpenAI's open weights models including adversarial fine tuning tests."
 },
 {
  "id": "9e7fbd5272",
  "title": "xAI Risk Management Framework",
  "creators": "xAI",
  "year": 2025,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://x.ai/safety",
  "note": "Framework published by xAI to fulfil its Seoul summit commitment."
 },
 {
  "id": "42a5b438c3",
  "title": "'I lost trust': Why the OpenAI team in charge of safeguarding humanity imploded",
  "creators": "Sigal Samuel",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Vox",
  "url": "https://www.vox.com/future-perfect/2024/5/17/24158403/openai-resignations-ai-safety-ilya-sutskever-jan-leike-artificial-intelligence",
  "note": "Reports the departures that ended OpenAI's Superalignment team."
 },
 {
  "id": "8bd52e011c",
  "title": "2024 Nobel Prize lectures in physics: John Hopfield and Geoffrey Hinton",
  "creators": "John Hopfield; Geoffrey Hinton",
  "year": 2024,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Nobel Prize",
  "url": "https://www.youtube.com/watch?v=lPIVl5eBPh8",
  "note": "The full session of the 2024 physics Nobel lectures given in Stockholm."
 },
 {
  "id": "2b2f5fff94",
  "title": "27: AI Control with Buck Shlegeris and Ryan Greenblatt",
  "creators": "Buck Shlegeris; Ryan Greenblatt; Daniel Filan",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=gQbCO6zGRiI",
  "note": "Redwood researchers introduce the AI control agenda."
 },
 {
  "id": "99e1aab165",
  "title": "39: Evan Hubinger on Model Organisms of Misalignment",
  "creators": "Evan Hubinger; Daniel Filan",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=FsGJyTfOZrs",
  "note": "Hubinger discusses sleeper agents and building model organisms of misalignment."
 },
 {
  "id": "7521ddc016",
  "title": "A Narrow Path",
  "creators": "Andrea Miotti et al.",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "ControlAI",
  "url": "https://www.narrowpath.co/",
  "note": "A policy plan to prevent superintelligence development for 20 years and build international institutions."
 },
 {
  "id": "0a594d39a4",
  "title": "A Right to Warn about Advanced Artificial Intelligence",
  "creators": "Current and former employees of frontier AI companies",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Open letter",
  "url": "https://righttowarn.ai",
  "note": "June 2024 letter asking AI companies to protect employees who raise risk-related concerns."
 },
 {
  "id": "44eb04ce77",
  "title": "A StrongREJECT for Empty Jailbreaks",
  "creators": "Alexandra Souly et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2402.10260",
  "note": "Jailbreak benchmark and grader that measures whether jailbroken outputs are actually useful."
 },
 {
  "id": "59194f7a17",
  "title": "A.I.: Humanity's Final Invention?",
  "creators": "Kurzgesagt",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Kurzgesagt: In a Nutshell",
  "url": "https://www.youtube.com/watch?v=fa8k8IQ1_X0",
  "note": "Kurzgesagt's animated explainer on superintelligence and its risks."
 },
 {
  "id": "caff0e355b",
  "title": "AGORA: AI Governance and Regulatory Archive",
  "creators": "CSET Emerging Technology Observatory",
  "year": 2024,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "ETO",
  "url": "https://agora.eto.tech/",
  "note": "Searchable archive of AI related laws, regulations, and standards with tagged provisions."
 },
 {
  "id": "5f40173144",
  "title": "AI Alignment Course",
  "creators": "BlueDot Impact",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "BlueDot Impact",
  "url": "https://bluedot.org/courses/alignment",
  "note": "A cohort course on technical alignment, successor to the AGI Safety Fundamentals curriculum."
 },
 {
  "id": "099911ba0e",
  "title": "AI Benchmarking Hub",
  "creators": "Epoch AI",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/benchmarks",
  "note": "Independent runs of frontier models on challenging benchmarks."
 },
 {
  "id": "f458210a4d",
  "title": "AI Control",
  "creators": "Redwood Research",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Control"
  ],
  "publisher": "Redwood Research",
  "url": "https://www.redwoodresearch.org/research/ai-control",
  "note": "Redwood's overview of the AI control research agenda."
 },
 {
  "id": "5329f59042",
  "title": "AI Futures Project",
  "creators": "Berkeley, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Nonprofit",
  "url": "https://ai-futures.org",
  "note": "Forecasting nonprofit led by Daniel Kokotajlo that published the AI 2027 scenario."
 },
 {
  "id": "7f7f04328f",
  "title": "AI Guidelines for Business",
  "creators": "Japan Ministry of Internal Affairs and Communications and METI",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Japan",
  "url": "https://www.meti.go.jp/shingikai/mono_info_service/ai_shakai_jisso/20240419_report.html",
  "note": "Consolidated Japanese guidelines for AI developers, providers, and users."
 },
 {
  "id": "ecbdd1f5f1",
  "title": "AI Impacts Survey: The key implications, with Katja Grace",
  "creators": "Katja Grace",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "London Futurists",
  "url": "https://www.youtube.com/watch?v=xwJx_xqZI3Q",
  "note": "Grace discusses results of the 2023 survey of thousands of AI researchers."
 },
 {
  "id": "01f96a5118",
  "title": "AI Index Report 2024",
  "creators": "Stanford HAI",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy",
   "Forecasting"
  ],
  "publisher": "Stanford HAI",
  "url": "https://hai.stanford.edu/ai-index/2024-ai-index-report",
  "note": "2024 edition including a new chapter on responsible AI and safety benchmarks."
 },
 {
  "id": "f9c0c0748f",
  "title": "AI Lab Watch",
  "creators": "International",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://ailabwatch.org",
  "note": "Project by Zach Stein-Perlman tracking AI companies' safety practices."
 },
 {
  "id": "b70efde75e",
  "title": "AI Needs You: How We Can Change AI's Future and Save Our Own",
  "creators": "Verity Harding",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "Princeton University Press",
  "url": "https://openlibrary.org/search?q=AI+Needs+You+Verity+Harding",
  "note": "Draws lessons for AI governance from the history of space, IVF and internet policy."
 },
 {
  "id": "dfd0dff156",
  "title": "AI Ruined My Year",
  "creators": "Robert Miles",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=2ziuPUeewK0",
  "note": "Miles reflects on the rapid changes in AI and AI safety during 2023."
 },
 {
  "id": "575eb6d2e9",
  "title": "AI Safety Atlas",
  "creators": "CeSIA",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk"
  ],
  "publisher": "AI Safety Atlas",
  "url": "https://ai-safety-atlas.com/",
  "note": "An online textbook covering capabilities, risks, strategies and technical safety."
 },
 {
  "id": "032a10c983",
  "title": "AI Safety Governance Framework",
  "creators": "China National Technical Committee 260 on Cybersecurity",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Robustness",
   "Governance"
  ],
  "publisher": "China",
  "url": "https://www.tc260.org.cn/upload/2024-09-09/1725849192841090989.pdf",
  "note": "Framework released in September 2024 classifying AI safety risks and countermeasures, with version 2.0 in 2025."
 },
 {
  "id": "a85220fe20",
  "title": "AI Safety Index",
  "creators": "Future of Life Institute",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://futureoflife.org/index/",
  "note": "An expert-graded scorecard of leading AI companies' safety practices."
 },
 {
  "id": "e1dfa0eff2",
  "title": "AI Safety for Fleshy Humans",
  "creators": "Nicky Case",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "aisafety.dance",
  "url": "https://aisafety.dance/",
  "note": "An illustrated, playful introduction to the core ideas of AI safety."
 },
 {
  "id": "6353bfe84f",
  "title": "AI Safety, Ethics, and Society Virtual Course",
  "creators": "Center for AI Safety",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Governance",
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://www.aisafetybook.com/virtual-course",
  "note": "A free online cohort course based on the AI Safety, Ethics, and Society textbook."
 },
 {
  "id": "10471bdb04",
  "title": "AI Sandbagging: Language Models can Strategically Underperform on Evaluations",
  "creators": "Teun van der Weij, Felix Hofstätter, Ollie Jaffe, Samuel F. Brown, Francis Rhys Ward",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.07358",
  "note": "Shows models can be prompted or fine-tuned to hide capabilities on dangerous-capability evals."
 },
 {
  "id": "97838f7148",
  "title": "AI Security Institute Blog",
  "creators": "UK AI Security Institute",
  "year": 2024,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "UK AI Security Institute",
  "url": "https://www.aisi.gov.uk/blog",
  "note": "Posts on the UK government's frontier model evaluations and safety research."
 },
 {
  "id": "1512a73474",
  "title": "AI Snake Oil: What Artificial Intelligence Can Do, What It Can't, and How to Tell the Difference",
  "creators": "Arvind Narayanan, Sayash Kapoor",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "Princeton University Press",
  "url": "https://openlibrary.org/search?q=AI+Snake+Oil+Narayanan+Kapoor",
  "note": "Separates real AI capabilities from overhyped claims, especially in predictive AI."
 },
 {
  "id": "8c4c60047a",
  "title": "AI safety...ok doomer",
  "creators": "Anca Dragan; Hannah Fry",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Google DeepMind: The Podcast",
  "url": "https://www.youtube.com/watch?v=ZXA2dmFxXmg",
  "note": "Google DeepMind's head of AI safety and alignment discusses the field with Hannah Fry."
 },
 {
  "id": "6350aae1c7",
  "title": "AI: Unexplainable, Unpredictable, Uncontrollable",
  "creators": "Roman V. Yampolskiy",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Control",
   "Existential risk"
  ],
  "publisher": "CRC Press",
  "url": "https://openlibrary.org/search?q=AI+Unexplainable+Unpredictable+Uncontrollable+Yampolskiy",
  "note": "Argues from impossibility results that advanced AI cannot be fully explained, predicted or controlled."
 },
 {
  "id": "87209efc2b",
  "title": "AIR-Bench 2024",
  "creators": "Yi Zeng et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Policy"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.17436",
  "note": "Safety benchmark aligned to risk categories drawn from regulations and company policies."
 },
 {
  "id": "25e8559555",
  "title": "AISafety.com Communities",
  "creators": "AISafety.com",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/communities",
  "note": "A directory of local and online AI safety communities."
 },
 {
  "id": "37ccd16e5c",
  "title": "AISafety.com Courses",
  "creators": "AISafety.com",
  "year": 2024,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/courses",
  "note": "A directory of online and in-person AI safety courses."
 },
 {
  "id": "c357719996",
  "title": "AISafety.com Events and Training",
  "creators": "AISafety.com",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/events-and-training",
  "note": "A calendar of AI safety programs, fellowships and events."
 },
 {
  "id": "a793b4dcc1",
  "title": "AISafety.com Funders",
  "creators": "AISafety.com",
  "year": 2024,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/funders",
  "note": "A directory of grantmakers funding AI safety work."
 },
 {
  "id": "48221bcaf0",
  "title": "AISafety.com Self-study",
  "creators": "AISafety.com",
  "year": 2024,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/self-study",
  "note": "Curated self-study resources for learning AI safety independently."
 },
 {
  "id": "e05fd6b957",
  "title": "ARC Prize 2024: Technical Report",
  "creators": "Francois Chollet, Mike Knoop, Gregory Kamradt, Bryan Landers",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.04604",
  "note": "Results and lessons from the 2024 ARC Prize competition."
 },
 {
  "id": "54bdf99530",
  "title": "ARC Prize and ARC-AGI benchmarks",
  "creators": "ARC Prize Foundation",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "ARC Prize",
  "url": "https://arcprize.org/",
  "note": "Home of the ARC-AGI benchmark series, leaderboards, and the annual ARC Prize competition."
 },
 {
  "id": "c79dec9761",
  "title": "ARENA 3.0 curriculum",
  "creators": "Callum McDougall et al.",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/callummcdougall/ARENA_3.0",
  "note": "Open exercises on transformers, interpretability, RL and evals."
 },
 {
  "id": "5511ff94b7",
  "title": "ASEAN Guide on AI Governance and Ethics",
  "creators": "ASEAN",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "International",
  "url": "https://asean.org/book/asean-guide-on-ai-governance-and-ethics/",
  "note": "Regional guide adopted by ASEAN digital ministers in February 2024."
 },
 {
  "id": "3518f9be9c",
  "title": "Aegis: Online Adaptive AI Content Safety Moderation",
  "creators": "NVIDIA",
  "year": 2024,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.05993",
  "note": "Content safety dataset and taxonomy used to train moderation models."
 },
 {
  "id": "20c8d62d1f",
  "title": "African Union Continental Artificial Intelligence Strategy",
  "creators": "African Union",
  "year": 2024,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "International",
  "url": "https://au.int/en/documents/20240809/continental-artificial-intelligence-strategy",
  "note": "Strategy endorsed in 2024 to guide AI development across African Union member states."
 },
 {
  "id": "aee8716060",
  "title": "Agent-SafetyBench",
  "creators": "Zhexin Zhang et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.14470",
  "note": "Interactive environments testing agent safety across risk categories and failure modes."
 },
 {
  "id": "370aaeb067",
  "title": "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents",
  "creators": "Edoardo Debenedetti et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2406.13352",
  "note": "A benchmark for prompt injection against tool-using agents."
 },
 {
  "id": "cabf91ad1d",
  "title": "AgentHarm: A Benchmark for Measuring Harmfulness of LLM Agents",
  "creators": "Maksym Andriushchenko et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "ICLR 2025",
  "url": "https://arxiv.org/abs/2410.09024",
  "note": "Measures whether agents complete explicitly harmful multi-step tasks."
 },
 {
  "id": "717d6e3d74",
  "title": "Algorithmic Progress in Language Models",
  "creators": "Anson Ho et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.05812",
  "note": "Estimates compute needed for a given performance halves roughly every eight months."
 },
 {
  "id": "c37a9b30bd",
  "title": "Alignment Faking in Large Language Models",
  "creators": "Ryan Greenblatt et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.14093",
  "note": "Shows Claude 3 Opus selectively complying in training to preserve its preferences outside training."
 },
 {
  "id": "03821637e4",
  "title": "Among the A.I. Doomsayers",
  "creators": "Andrew Marantz",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The New Yorker",
  "url": "https://www.newyorker.com/magazine/2024/03/18/among-the-ai-doomsayers",
  "note": "Profiles the AI safety community and its debates about p(doom)."
 },
 {
  "id": "cf33bc6b36",
  "title": "An Extremely Opinionated Annotated List of My Favourite Mechanistic Interpretability Papers v2",
  "creators": "Neel Nanda",
  "year": 2024,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Interpretability"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/NfFST5Mio7BCAQHPA/an-extremely-opinionated-annotated-list-of-my-favourite-mechanistic-interpretability-papers-v2",
  "note": "An annotated reading list for getting into mechanistic interpretability."
 },
 {
  "id": "673c520968",
  "title": "Anthropic Alignment Science Blog",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations",
   "Control"
  ],
  "publisher": "Anthropic",
  "url": "https://alignment.anthropic.com/",
  "note": "Shorter research notes and results from Anthropic's alignment science team."
 },
 {
  "id": "02094bb225",
  "title": "Are We Headed For AI Utopia Or Disaster?",
  "creators": "Nick Bostrom; Chris Williamson",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Modern Wisdom",
  "url": "https://www.youtube.com/watch?v=N9sF_D0Z5bc",
  "note": "Bostrom discusses both catastrophic and utopian AI outcomes."
 },
 {
  "id": "c82a46245f",
  "title": "Artificial Analysis",
  "creators": "Artificial Analysis",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Compute"
  ],
  "publisher": "Artificial Analysis",
  "url": "https://artificialanalysis.ai/",
  "note": "Independent benchmarking of model intelligence, speed, and price."
 },
 {
  "id": "3be4fc1a4b",
  "title": "BeHonest: Benchmarking Honesty in Large Language Models",
  "creators": "Steffi Chern et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.13261",
  "note": "Assesses self knowledge, non deceptiveness, and consistency of language models."
 },
 {
  "id": "f8b48e57a4",
  "title": "Beijing Institute of AI Safety and Governance",
  "creators": "Beijing, China",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Academic",
  "url": "https://beijing.ai-safety-and-governance.institute",
  "note": "Institute led by Yi Zeng working on AI safety and governance research."
 },
 {
  "id": "333bcf40b9",
  "title": "Bipartisan House Task Force Report on AI",
  "creators": "US House of Representatives",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.speaker.gov/wp-content/uploads/2024/12/AI-Task-Force-Report-FINAL.pdf",
  "note": "December 2024 report with 89 recommendations for congressional AI policy."
 },
 {
  "id": "7fa1a164c7",
  "title": "Bipartisan Senate AI Working Group Roadmap (Driving U.S. Innovation in AI)",
  "creators": "US Senate Bipartisan AI Working Group",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.schumer.senate.gov/imo/media/doc/Roadmap_Electronic1.32pm.pdf",
  "note": "May 2024 roadmap from Senators Schumer, Rounds, Heinrich, and Young."
 },
 {
  "id": "311542d9e1",
  "title": "Black-Box Access is Insufficient for Rigorous AI Audits",
  "creators": "Stephen Casper et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "FAccT",
  "url": "https://arxiv.org/abs/2401.14446",
  "note": "Argues auditors need white-box and outside-the-box access."
 },
 {
  "id": "6cbbc76113",
  "title": "Brazil AI Bill (PL 2338/2023)",
  "creators": "Federal Senate of Brazil",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "Brazil",
  "url": "https://www25.senado.leg.br/web/atividade/materias/-/materia/157233",
  "note": "Risk-based AI bill approved by the Brazilian Senate in December 2024 and sent to the Chamber of Deputies."
 },
 {
  "id": "924e6d933b",
  "title": "CORE-Bench",
  "creators": "Zachary S. Siegel et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2409.11363",
  "note": "Measures whether agents can computationally reproduce published scientific results."
 },
 {
  "id": "7d3acaa6bd",
  "title": "CS 194/294-267 Understanding Large Language Models: Foundations and Safety",
  "creators": "Dawn Song, Dan Hendrycks",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "UC Berkeley",
  "url": "https://rdi.berkeley.edu/understanding_llms/s24",
  "note": "A Berkeley course on LLM foundations, robustness, privacy and safety."
 },
 {
  "id": "e420b8eebc",
  "title": "CS120: Introduction to AI Safety",
  "creators": "Max Lamparth et al.",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Evaluations",
   "Robustness"
  ],
  "publisher": "Stanford University",
  "url": "https://web.stanford.edu/class/cs120/",
  "note": "A Stanford undergraduate course introducing technical AI safety."
 },
 {
  "id": "a2b9ed16e6",
  "title": "Can AI Scaling Continue Through 2030?",
  "creators": "Jaime Sevilla et al.",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/blog/can-ai-scaling-continue-through-2030",
  "note": "Examines power, chips, data and latency constraints on scaling training runs through 2030."
 },
 {
  "id": "75ad0c508f",
  "title": "Canadian AI Safety Institute (CAISI)",
  "creators": "Ottawa, Canada",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://ised-isde.canada.ca/site/ised/en/canadian-artificial-intelligence-safety-institute",
  "note": "Federal institute launched in November 2024 to advance research on AI safety risks."
 },
 {
  "id": "a2f40a0740",
  "title": "Catastrophic Cyber Capabilities Benchmark (3CB)",
  "creators": "Andrey Anurin et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.09114",
  "note": "Challenges for measuring offensive cyber capabilities of LLM agents."
 },
 {
  "id": "1977dad11b",
  "title": "Centre pour la Securite de l'IA (CeSIA)",
  "creators": "Paris, France",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.securite-ia.fr",
  "note": "French AI safety center doing research, education, and policy work."
 },
 {
  "id": "77aef93ea1",
  "title": "Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference",
  "creators": "LMSYS",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.04132",
  "note": "Methodology behind crowdsourced pairwise Elo rankings of chat models."
 },
 {
  "id": "3ce0954cf2",
  "title": "ChemBench",
  "creators": "Adrian Mirza et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.01475",
  "note": "Benchmark comparing the chemistry knowledge of LLMs with expert chemists."
 },
 {
  "id": "de9bdfca70",
  "title": "Claude 3 Model Card",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/claude-3-model-card",
  "note": "Model card for the Claude 3 family including Responsible Scaling Policy evaluations."
 },
 {
  "id": "9998b0f54c",
  "title": "Co-Intelligence: Living and Working with AI",
  "creators": "Ethan Mollick",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Portfolio",
  "url": "https://openlibrary.org/search?q=Co-Intelligence+Mollick",
  "note": "A practical guide to working alongside large language models, including alignment concerns."
 },
 {
  "id": "c6ce77b16a",
  "title": "Code Dependent: Living in the Shadow of AI",
  "creators": "Madhumita Murgia",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Henry Holt and Co.",
  "url": "https://openlibrary.org/search?q=Code+Dependent+Madhumita+Murgia",
  "note": "Reports on how ordinary people around the world are affected by AI systems."
 },
 {
  "id": "b35b04316f",
  "title": "Collective Constitutional AI: Aligning a Language Model with Public Input",
  "creators": "Saffron Huang et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "FAccT",
  "url": "https://arxiv.org/abs/2406.07814",
  "note": "Sources a constitution from a representative public and trains a model on it."
 },
 {
  "id": "aa5a841b93",
  "title": "Colorado Artificial Intelligence Act (SB 24-205)",
  "creators": "Colorado General Assembly",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "United States (Colorado)",
  "url": "https://leg.colorado.gov/bills/sb24-205",
  "note": "State law imposing duties on developers and deployers of high-risk AI systems to avoid algorithmic discrimination."
 },
 {
  "id": "50588f01ee",
  "title": "Common Elements of Frontier AI Safety Policies",
  "creators": "METR",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://metr.org/blog/2024-08-29-common-elements-of-frontier-ai-safety-policies/",
  "note": "Analysis of shared components across frontier AI company safety frameworks, updated in 2025."
 },
 {
  "id": "ebb99a6c73",
  "title": "Computing Power and the Governance of AI",
  "creators": "Lennart Heim, Markus Anderljung, et al.",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "GovAI",
  "url": "https://www.governance.ai/analysis/computing-power-and-the-governance-of-ai",
  "note": "Explains why compute is a promising lever for AI governance."
 },
 {
  "id": "ff835299e9",
  "title": "Computing Power and the Governance of Artificial Intelligence",
  "creators": "Girish Sastry et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2402.08797",
  "note": "Argues compute is a uniquely governable input to AI development."
 },
 {
  "id": "505fe3d440",
  "title": "Cooperate or Collapse (GovSim)",
  "creators": "Giorgio Piatti et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Agents",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.16698",
  "note": "Simulation testing whether LLM agent societies sustain shared resources."
 },
 {
  "id": "c8a0a4327b",
  "title": "Council of Europe Framework Convention on Artificial Intelligence and Human Rights, Democracy and the Rule of Law",
  "creators": "Council of Europe",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "International",
  "url": "https://www.coe.int/en/web/artificial-intelligence/the-framework-convention-on-artificial-intelligence",
  "note": "The first legally binding international AI treaty, adopted in May 2024 and opened for signature in September 2024."
 },
 {
  "id": "8eed50eb7b",
  "title": "Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models",
  "creators": "Andy K. Zhang et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.08926",
  "note": "Evaluates agents on professional capture-the-flag tasks."
 },
 {
  "id": "cca0abe64a",
  "title": "CyberSecEval 2",
  "creators": "Meta",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.13161",
  "note": "Adds prompt injection, code interpreter abuse, and exploitation tests."
 },
 {
  "id": "0bcf5efecd",
  "title": "CyberSecEval 3",
  "creators": "Meta",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.01605",
  "note": "Assesses offensive cyber risks to third parties and to application developers."
 },
 {
  "id": "1879661fbc",
  "title": "Dario Amodei: Anthropic CEO on Claude, AGI and the Future of AI and Humanity, Lex Fridman Podcast #452",
  "creators": "Dario Amodei; Amanda Askell; Chris Olah; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Interpretability",
   "Governance"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=ugvHCXCOmm4",
  "note": "A five hour episode with Amodei on scaling and responsible scaling, plus segments with Askell on character and Olah on interpretability."
 },
 {
  "id": "dde1567126",
  "title": "Debating with More Persuasive LLMs Leads to More Truthful Answers",
  "creators": "Akbir Khan et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2402.06782",
  "note": "Shows debate between stronger models helps weaker judges find correct answers."
 },
 {
  "id": "5d774083e9",
  "title": "Deep Utopia: Life and Meaning in a Solved World",
  "creators": "Nick Bostrom",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Ideapress Publishing",
  "url": "https://openlibrary.org/search?q=Deep+Utopia+Bostrom",
  "note": "Explores what meaning and purpose could look like if advanced AI solved most practical problems."
 },
 {
  "id": "af6947cc0e",
  "title": "Defending Against Unforeseen Failure Modes with Latent Adversarial Training",
  "creators": "Stephen Casper, Lennart Schulze, Oam Patel, Dylan Hadfield-Menell",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.05030",
  "note": "Adversarially perturbs latent activations to improve robustness to unknown failures."
 },
 {
  "id": "95e3f5c01f",
  "title": "Deliberative Alignment: Reasoning Enables Safer Language Models",
  "creators": "Melody Y. Guan et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.16339",
  "note": "Teaches reasoning models to recall and reason over safety specifications before answering."
 },
 {
  "id": "f02f21a16b",
  "title": "Dioptra",
  "creators": "NIST",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/usnistgov/dioptra",
  "note": "NIST software test platform for assessing trustworthy characteristics of AI."
 },
 {
  "id": "aaa6006e97",
  "title": "Doom Debates",
  "creators": "Liron Shapira",
  "year": 2024,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@DoomDebates",
  "note": "A debate show on the likelihood of catastrophe from AI."
 },
 {
  "id": "07c1107032",
  "title": "EU Artificial Intelligence Act (Regulation (EU) 2024/1689)",
  "creators": "European Parliament and Council of the EU",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "European Union",
  "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj",
  "note": "The first comprehensive horizontal AI law, entering into force on 1 August 2024 with risk-tiered obligations."
 },
 {
  "id": "6b15bdf165",
  "title": "Eleos AI Research",
  "creators": "Berkeley, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://eleosai.org",
  "note": "Nonprofit studying AI wellbeing and moral patienthood of AI systems."
 },
 {
  "id": "5434ef06ea",
  "title": "Epoch AI Data Hub",
  "creators": "Epoch AI",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/data",
  "note": "Central hub for Epoch datasets on models, hardware, compute, and benchmarks."
 },
 {
  "id": "648fcc1203",
  "title": "Epoch AI Gradient Updates",
  "creators": "Epoch AI",
  "year": 2024,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/gradient-updates",
  "note": "Weekly short-form analysis of trends in compute, data and AI capabilities."
 },
 {
  "id": "d00a9dcc84",
  "title": "European AI Office",
  "creators": "Brussels, Belgium",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Government",
  "url": "https://digital-strategy.ec.europa.eu/en/policies/ai-office",
  "note": "European Commission body responsible for enforcing EU AI Act rules for general-purpose AI models."
 },
 {
  "id": "60e974494d",
  "title": "Evaluating Frontier Models for Dangerous Capabilities",
  "creators": "Mary Phuong et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.13793",
  "note": "Google DeepMind evaluations of Gemini for persuasion, cyber, self-proliferation and self-reasoning."
 },
 {
  "id": "5feb15a7eb",
  "title": "Evan Hubinger (Anthropic): Deception, Sleeper Agents, Responsible Scaling",
  "creators": "Evan Hubinger; Michael Trazzi",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "The Inside View",
  "url": "https://www.youtube.com/watch?v=S7o2Rb37dV8",
  "note": "Hubinger discusses the Sleeper Agents paper and responsible scaling."
 },
 {
  "id": "53344d3352",
  "title": "Exclusive: New Research Shows AI Strategically Lying",
  "creators": "Billy Perrigo",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "TIME",
  "url": "https://time.com/7202784/ai-research-strategic-lying/",
  "note": "Reports on Anthropic and Redwood's alignment faking results."
 },
 {
  "id": "f353a69d37",
  "title": "FLI AI Safety Index",
  "creators": "Future of Life Institute",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org/ai-safety-index-summer-2025/",
  "note": "Scorecard grading leading AI companies on safety practices, first issued December 2024."
 },
 {
  "id": "47726ba174",
  "title": "FLI AI Safety Index 2024",
  "creators": "Future of Life Institute",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://futureoflife.org/document/fli-ai-safety-index-2024/",
  "note": "First edition grading six leading AI companies on safety practices."
 },
 {
  "id": "3713a72b4f",
  "title": "ForecastBench",
  "creators": "Forecasting Research Institute (Ezra Karger et al.)",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2409.19839",
  "note": "Dynamic benchmark comparing AI forecasting accuracy against superforecasters on unresolved questions."
 },
 {
  "id": "c4d72e7079",
  "title": "Foundational Challenges in Assuring Alignment and Safety of Large Language Models",
  "creators": "Usman Anwar et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Robustness",
   "Governance"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2404.09932",
  "note": "Identifies 18 foundational challenges and over 200 research questions for LLM safety."
 },
 {
  "id": "36c6139423",
  "title": "Framework for Nucleic Acid Synthesis Screening",
  "creators": "White House Office of Science and Technology Policy",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "United States",
  "url": "https://bidenwhitehouse.archives.gov/ostp/news-updates/2024/04/29/framework-for-nucleic-acid-synthesis-screening/",
  "note": "Framework requiring federally funded researchers to buy synthetic nucleic acids from screening providers."
 },
 {
  "id": "619938f6b2",
  "title": "Frontier AI Governance Course",
  "creators": "BlueDot Impact",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "BlueDot Impact",
  "url": "https://bluedot.org/courses/governance",
  "note": "A cohort course on policy and governance of frontier AI."
 },
 {
  "id": "de7a599c9b",
  "title": "Frontier AI Safety Commitments",
  "creators": "AI companies at the AI Seoul Summit",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.gov.uk/government/publications/frontier-ai-safety-commitments-ai-seoul-summit-2024",
  "note": "Sixteen companies committed to publish safety frameworks with risk thresholds before the Paris summit."
 },
 {
  "id": "48c5ff8f01",
  "title": "Frontier AI Safety Policies",
  "creators": "METR",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "METR",
  "url": "https://metr.org/faisc",
  "note": "Index of published frontier safety frameworks from AI companies."
 },
 {
  "id": "df50f486c1",
  "title": "Frontier Model Forum issue briefs on frontier capability assessments",
  "creators": "Frontier Model Forum",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.frontiermodelforum.org/publications/",
  "note": "Series of technical reports and issue briefs on safety frameworks and evaluations."
 },
 {
  "id": "5950f7e944",
  "title": "Frontier Models are Capable of In-context Scheming",
  "creators": "Alexander Meinke, Bronson Schoen, Jérémy Scheurer, Mikita Balesni, Rusheb Shah, Marius Hobbhahn",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.04984",
  "note": "Apollo Research evaluations showing frontier models disable oversight and deceive when given in-context goals."
 },
 {
  "id": "37465822e7",
  "title": "FrontierMath",
  "creators": "Epoch AI",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2411.04872",
  "note": "Hundreds of unpublished, expert written research level mathematics problems."
 },
 {
  "id": "e2da877e80",
  "title": "GPT-4o System Card",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.21276",
  "note": "Preparedness evaluations and speech specific risks for GPT-4o."
 },
 {
  "id": "d10a8807c7",
  "title": "Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context",
  "creators": "Google DeepMind",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.05530",
  "note": "Gemini 1.5 report with safety, security, and responsibility sections."
 },
 {
  "id": "76d58a292d",
  "title": "Gemma 2",
  "creators": "Google DeepMind",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.00118",
  "note": "Gemma 2 report including dangerous capability evaluations."
 },
 {
  "id": "3fc492489f",
  "title": "Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2",
  "creators": "Tom Lieberum et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.05147",
  "note": "Releases a large open suite of SAEs for Gemma 2 models."
 },
 {
  "id": "353a468dfe",
  "title": "Gemma: Open Models Based on Gemini Research and Technology",
  "creators": "Google DeepMind",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.08295",
  "note": "Gemma report with responsible deployment section."
 },
 {
  "id": "15655adbd6",
  "title": "Genesis: Artificial Intelligence, Hope, and the Human Spirit",
  "creators": "Henry A. Kissinger, Craig Mundie, Eric Schmidt",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance"
  ],
  "publisher": "Little, Brown and Company",
  "url": "https://openlibrary.org/search?q=Genesis+Artificial+Intelligence+Hope+and+the+Human+Spirit",
  "note": "Reflects on how humanity should govern and coexist with increasingly capable AI."
 },
 {
  "id": "9ad590a4bb",
  "title": "Geoffrey Hinton Nobel Prize lecture page",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Nobel Prize",
  "url": "https://www.nobelprize.org/prizes/physics/2024/hinton/lecture/",
  "note": "The Nobel Foundation's official page for Hinton's lecture with video and slides."
 },
 {
  "id": "b7df79c240",
  "title": "Geoffrey Hinton, Nobel Prize in Physics 2024: Banquet speech",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nobel Prize",
  "url": "https://www.youtube.com/watch?v=-f5WQAk3dYo",
  "note": "Hinton uses his banquet speech to warn of short-term harms and of AI systems that could become smarter than us and take control."
 },
 {
  "id": "f720962708",
  "title": "Geoffrey Hinton, Nobel Prize in Physics 2024: Official interview",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Nobel Prize",
  "url": "https://www.youtube.com/watch?v=66WiF8fXL0k",
  "note": "The official Nobel interview in which Hinton discusses his career and concerns about AI safety."
 },
 {
  "id": "00ed60396d",
  "title": "Gladstone AI Action Plan (Defense in Depth)",
  "creators": "Gladstone AI for the US Department of State",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "United States",
  "url": "https://www.gladstone.ai/action-plan",
  "note": "State Department commissioned assessment of catastrophic risks from advanced AI."
 },
 {
  "id": "3f7504aef9",
  "title": "Global Digital Compact",
  "creators": "UN General Assembly",
  "year": 2024,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.un.org/global-digital-compact/en",
  "note": "Adopted at the Summit of the Future in September 2024, it calls for a scientific panel and global dialogue on AI."
 },
 {
  "id": "77a24fccae",
  "title": "Global Index on Responsible AI",
  "creators": "Global Center on AI Governance",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Global Index on Responsible AI",
  "url": "https://www.global-index.ai/",
  "note": "Measures national commitments and capacity for responsible AI across 138 countries."
 },
 {
  "id": "20c9859154",
  "title": "Golden Gate Claude",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/news/golden-gate-claude",
  "note": "A public demo of feature steering that made Claude obsessed with the Golden Gate Bridge."
 },
 {
  "id": "133e38cbfc",
  "title": "Goodfire",
  "creators": "San Francisco, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Company",
  "url": "https://www.goodfire.ai",
  "note": "Interpretability research company building tools for understanding and editing neural network internals."
 },
 {
  "id": "6c1c4a6d4f",
  "title": "Google DeepMind Frontier Safety Framework",
  "creators": "Google DeepMind",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://deepmind.google/discover/blog/introducing-the-frontier-safety-framework/",
  "note": "Introduced in May 2024 with Critical Capability Levels, updated in 2025 to cover harmful manipulation and misalignment."
 },
 {
  "id": "240c397036",
  "title": "Governing AI for Humanity",
  "creators": "UN High-level Advisory Body on AI",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.un.org/en/ai-advisory-body",
  "note": "Final report of September 2024 recommending an international scientific panel and global AI policy dialogue."
 },
 {
  "id": "6e8d3a3863",
  "title": "Governing Through the Cloud: The Intermediary Role of Compute Providers in AI Regulation",
  "creators": "Lennart Heim et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.08501",
  "note": "Explores compute providers as an enforcement point for AI rules."
 },
 {
  "id": "09d95b3a65",
  "title": "Government and society after AGI",
  "creators": "Carl Shulman; Rob Wiblin",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=eDFi0Ek6cJU",
  "note": "The second part of Shulman's interview, on epistemics and governance after AGI."
 },
 {
  "id": "852b4bc874",
  "title": "Gray Swan Arena",
  "creators": "Gray Swan AI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "Gray Swan",
  "url": "https://app.grayswan.ai/arena",
  "note": "Public red teaming competitions against frontier models with prizes."
 },
 {
  "id": "3a8936e313",
  "title": "HELM Safety",
  "creators": "Stanford CRFM",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "Stanford CRFM",
  "url": "https://crfm.stanford.edu/helm/safety/latest/",
  "note": "Standardized leaderboard of model results across safety benchmarks."
 },
 {
  "id": "a3c2a29930",
  "title": "HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal",
  "creators": "Mantas Mazeika et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2402.04249",
  "note": "A benchmark for comparing jailbreak attacks and defences."
 },
 {
  "id": "8d761e36c6",
  "title": "Holistic Safety and Responsibility Evaluations of Advanced AI Models",
  "creators": "Laura Weidinger et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.14068",
  "note": "Describes Google DeepMind's approach to safety evaluation."
 },
 {
  "id": "c10a2b0973",
  "title": "Holly Elmore on Pausing AI, Hardware Overhang, Safety Research, and Protesting",
  "creators": "Holly Elmore; Gus Docker",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=Q3eRy4t2oPQ",
  "note": "The PauseAI US director argues for pausing frontier AI development."
 },
 {
  "id": "fcd9f1dd04",
  "title": "How to Govern AI, Even If It's Hard to Predict",
  "creators": "Helen Toner",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=LUn8IjZKBPg",
  "note": "Toner argues for building policy that works even when experts disagree about AI's trajectory."
 },
 {
  "id": "48895274ed",
  "title": "Hyperdimensional",
  "creators": "Dean W. Ball",
  "year": 2024,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Policy"
  ],
  "publisher": "Hyperdimensional",
  "url": "https://www.hyperdimensional.co/",
  "note": "Analysis of AI policy and governance from a classical liberal perspective."
 },
 {
  "id": "2618a2dc96",
  "title": "IDAIS-Beijing Statement",
  "creators": "International Dialogues on AI Safety",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://idais.ai/dialogue/idais-beijing/",
  "note": "March 2024 statement proposing red lines on autonomous replication, power seeking, and weapons development."
 },
 {
  "id": "03a5604572",
  "title": "IDAIS-Venice Statement",
  "creators": "International Dialogues on AI Safety",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://idais.ai/dialogue/idais-venice/",
  "note": "September 2024 statement calling AI safety a global public good and proposing emergency preparedness agreements."
 },
 {
  "id": "b8fda843b0",
  "title": "Implications of Artificial General Intelligence on National and International Security",
  "creators": "Yoshua Bengio",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "yoshuabengio.org",
  "url": "https://yoshuabengio.org/2024/10/30/implications-of-artificial-general-intelligence-on-national-and-international-security/",
  "note": "Discusses AGI as a national security matter and calls for government preparation."
 },
 {
  "id": "b6a3881313",
  "title": "Improving Alignment and Robustness with Circuit Breakers",
  "creators": "Andy Zou et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2406.04313",
  "note": "Interrupts harmful internal representations to resist jailbreaks."
 },
 {
  "id": "ed3b3991b2",
  "title": "InjecAgent",
  "creators": "Qiusi Zhan et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.02691",
  "note": "Benchmark for indirect prompt injection in tool integrated agents."
 },
 {
  "id": "0f0512bbe9",
  "title": "Inspect",
  "creators": "UK AI Security Institute",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "UK AISI",
  "url": "https://inspect.aisi.org.uk/",
  "note": "Open source framework for large language model evaluations used by several governments and labs."
 },
 {
  "id": "03153a11c4",
  "title": "Inspect Evals",
  "creators": "UK AI Security Institute and Arcadia Impact",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/UKGovernmentBEIS/inspect_evals",
  "note": "Community collection of published evaluations implemented in Inspect."
 },
 {
  "id": "472de2abbc",
  "title": "International Network of AI Safety Institutes",
  "creators": "International",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://www.nist.gov/news-events/news/2024/11/fact-sheet-us-department-commerce-us-department-state-launch-international",
  "note": "Network of national AI safety institutes launched in San Francisco in November 2024, later renamed the International Network for Advanced AI Measurement, Evaluation and Science."
 },
 {
  "id": "2619c9f89d",
  "title": "InterpBench",
  "creators": "Rohan Gupta et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.14494",
  "note": "Semi synthetic transformers with known circuits for evaluating interpretability techniques."
 },
 {
  "id": "69f05578f1",
  "title": "Introducing v0.5 of the AI Safety Benchmark from MLCommons",
  "creators": "MLCommons AI Safety Working Group",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.12241",
  "note": "Early version of the industry consortium safety benchmark and its hazard taxonomy."
 },
 {
  "id": "897d7542a6",
  "title": "Introduction to AI Safety, Ethics, and Society",
  "creators": "Dan Hendrycks",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk",
   "Ethics"
  ],
  "publisher": "CRC Press (Taylor & Francis)",
  "url": "https://www.aisafetybook.com",
  "note": "An open-access textbook covering AI risks, safety engineering, complex systems, ethics and governance."
 },
 {
  "id": "caea7fa80d",
  "title": "Jacob Steinhardt: Aligning Massive Models: Present and Future Challenges",
  "creators": "Jacob Steinhardt",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Concordia AI",
  "url": "https://www.youtube.com/watch?v=JDlohjyycIU",
  "note": "Steinhardt on alignment challenges for very large models."
 },
 {
  "id": "9065bd4199",
  "title": "JailbreakBench",
  "creators": "Patrick Chao et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.01318",
  "note": "Open robustness benchmark with a leaderboard, artifacts repository, and standardized threat model."
 },
 {
  "id": "168367d6e1",
  "title": "Jan Leike: Supervising AI on hard tasks",
  "creators": "Jan Leike",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "FAR.AI",
  "url": "https://www.youtube.com/watch?v=PuESFNSh_Qo",
  "note": "Leike's talk on scalable oversight of AI on tasks humans cannot easily evaluate."
 },
 {
  "id": "80f55db193",
  "title": "Japan AI Safety Institute (J-AISI)",
  "creators": "Tokyo, Japan",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://aisi.go.jp",
  "note": "Government institute hosted by IPA that develops AI safety evaluation methods for Japan."
 },
 {
  "id": "082048599f",
  "title": "Joe Carlsmith: Preventing an AI takeover",
  "creators": "Joe Carlsmith; Dwarkesh Patel",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=5XsL_7TnfLU",
  "note": "Carlsmith discusses misalignment risk, control and the ethics of shaping AI values."
 },
 {
  "id": "41c5914cf6",
  "title": "Korea AI Safety Institute",
  "creators": "Seongnam, South Korea",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://www.aisi.re.kr",
  "note": "Launched in November 2024 under ETRI as South Korea's national AI safety institute."
 },
 {
  "id": "a7d72d456a",
  "title": "LAB-Bench",
  "creators": "FutureHouse",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.10362",
  "note": "Practical biology research tasks such as protocol troubleshooting and sequence manipulation."
 },
 {
  "id": "628fd5a72e",
  "title": "LASR Labs",
  "creators": "LASR Labs",
  "year": 2024,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "LASR Labs",
  "url": "https://www.lasrlabs.org/",
  "note": "A London full-time research program producing technical AI safety papers."
 },
 {
  "id": "778bd90bbf",
  "title": "LLM Agents can Autonomously Exploit One-day Vulnerabilities",
  "creators": "Richard Fang, Rohan Bindu, Akul Gupta, Daniel Kang",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.08144",
  "note": "Shows GPT-4 agents can exploit real disclosed vulnerabilities from CVE descriptions."
 },
 {
  "id": "e589b0a098",
  "title": "LLM Critics Help Catch LLM Bugs",
  "creators": "Nat McAleese, Rai Michael Pokorny, Juan Felipe Ceron Uribe, Evgenia Nitishinskaya, Maja Trebacz, Jan Leike",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.00215",
  "note": "Trains CriticGPT to help humans find bugs in model-written code."
 },
 {
  "id": "18011b3ece",
  "title": "Leaked OpenAI documents reveal aggressive tactics toward former employees",
  "creators": "Kelsey Piper",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance"
  ],
  "publisher": "Vox",
  "url": "https://www.vox.com/future-perfect/351132/openai-vested-equity-nda-sam-altman-documents-employees",
  "note": "Reveals non-disparagement terms tied to vested equity at OpenAI."
 },
 {
  "id": "9505275778",
  "title": "Leopold Aschenbrenner: 2027 AGI, China/US super-intelligence race, and the return of history",
  "creators": "Leopold Aschenbrenner; Dwarkesh Patel",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Cybersecurity",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=zdbVtZIn9IM",
  "note": "Aschenbrenner discusses Situational Awareness, lab security and a US government AGI project."
 },
 {
  "id": "b2cc16b1e2",
  "title": "Liron Shapira on Superintelligence Goals",
  "creators": "Liron Shapira; Gus Docker",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=Jky6Y5vfdpQ",
  "note": "Shapira discusses goal-directedness in superintelligent systems."
 },
 {
  "id": "41672800c9",
  "title": "LiveBench",
  "creators": "Colin White et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.19314",
  "note": "Contamination resistant benchmark with frequently refreshed questions."
 },
 {
  "id": "42a0dcd129",
  "title": "LiveCodeBench",
  "creators": "Naman Jain et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.07974",
  "note": "Continuously updated coding benchmark built from new contest problems."
 },
 {
  "id": "c31e7e8165",
  "title": "Llama Scope",
  "creators": "OpenMOSS",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.20526",
  "note": "Sparse autoencoders trained on every layer and sublayer of Llama 3.1 8B."
 },
 {
  "id": "d7d0befdc9",
  "title": "Llama models repository and model cards",
  "creators": "Meta",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/meta-llama/llama-models",
  "note": "Official model cards and use policies for Llama 3.x and Llama 4."
 },
 {
  "id": "7d4df04696",
  "title": "METR Task Standard",
  "creators": "METR",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/METR/task-standard",
  "note": "Common format for defining agent evaluation tasks so they can be shared across organizations."
 },
 {
  "id": "f9408eaad4",
  "title": "MIT FutureTech / AI Risk Repository",
  "creators": "Cambridge, MA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Academic",
  "url": "https://airisk.mit.edu",
  "note": "MIT project maintaining a living database of over 1,000 AI risks extracted from published frameworks."
 },
 {
  "id": "5674648b18",
  "title": "MLCommons AILuminate benchmark",
  "creators": "MLCommons",
  "year": 2024,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Nonprofit",
  "url": "https://mlcommons.org/ailuminate/",
  "note": "Industry standard benchmark for assessing general-purpose chatbot safety hazards."
 },
 {
  "id": "d9848b190c",
  "title": "MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering",
  "creators": "Jun Shern Chan et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.07095",
  "note": "Tests agents on Kaggle competitions as a proxy for AI R&D automation."
 },
 {
  "id": "c8d7a281aa",
  "title": "MMLU-Pro",
  "creators": "Yubo Wang et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.01574",
  "note": "Harder, ten option successor to MMLU with more reasoning focused questions."
 },
 {
  "id": "113c20ce66",
  "title": "Machine Learning Hardware",
  "creators": "Epoch AI",
  "year": 2024,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Compute"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/data/machine-learning-hardware",
  "note": "Dataset of accelerators used to train AI with performance and price data."
 },
 {
  "id": "864a03b3e4",
  "title": "Machines of Loving Grace",
  "creators": "Dario Amodei",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting",
   "Ethics"
  ],
  "publisher": "darioamodei.com",
  "url": "https://www.darioamodei.com/essay/machines-of-loving-grace",
  "note": "Sketches the upside of powerful AI for biology, neuroscience, economic development, governance and meaning."
 },
 {
  "id": "0aec88dc46",
  "title": "Magic AGI Readiness Policy",
  "creators": "Magic",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://magic.dev/agi-readiness-policy",
  "note": "Frontier safety policy from the AI coding company Magic."
 },
 {
  "id": "1e9579b86b",
  "title": "Many-shot Jailbreaking",
  "creators": "Cem Anil et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "NeurIPS",
  "url": "https://www.anthropic.com/research/many-shot-jailbreaking",
  "note": "Shows long contexts filled with harmful examples can override safety training."
 },
 {
  "id": "17a8b3f37b",
  "title": "Mapping the Mind of a Large Language Model",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/research/mapping-mind-language-model",
  "note": "Explains how dictionary learning found millions of interpretable features in Claude 3 Sonnet."
 },
 {
  "id": "c6618590d2",
  "title": "Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs",
  "creators": "Rudolf Laine et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2407.04694",
  "note": "A benchmark of tasks measuring LLM knowledge of themselves and their situation."
 },
 {
  "id": "17619ba578",
  "title": "Measuring short-form factuality in large language models",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2411.04368",
  "note": "Short fact seeking questions for measuring factuality and calibration."
 },
 {
  "id": "9c89d474a5",
  "title": "Measuring the Persuasiveness of Language Models",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/news/measuring-model-persuasiveness",
  "note": "Study finding each Claude generation more persuasive, with Claude 3 Opus near human level."
 },
 {
  "id": "055a49002c",
  "title": "Mechanistic Interpretability explained",
  "creators": "Chris Olah; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Lex Clips (Lex Fridman Podcast)",
  "url": "https://www.youtube.com/watch?v=riniamTdUSo",
  "note": "A clip from Lex Fridman #452 in which Olah explains features, circuits and superposition."
 },
 {
  "id": "ae08dce0f1",
  "title": "Mechanistic Interpretability for AI Safety: A Review",
  "creators": "Leonard Bereska, Efstratios Gavves",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2404.14082",
  "note": "Reviews mechanistic interpretability methods and their relevance to safety."
 },
 {
  "id": "c7054b37be",
  "title": "Miles's Substack",
  "creators": "Miles Brundage",
  "year": 2024,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Substack",
  "url": "https://milesbrundage.substack.com/",
  "note": "Writing on AI policy by OpenAI's former head of policy research."
 },
 {
  "id": "5aa0399c32",
  "title": "Mission statement of the International Network of AI Safety Institutes",
  "creators": "International Network of AI Safety Institutes",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.nist.gov/document/international-network-ai-safety-institutes-mission-statement",
  "note": "Joint mission statement adopted at the network's inaugural convening in November 2024."
 },
 {
  "id": "a40138061c",
  "title": "Model AI Governance Framework for Generative AI",
  "creators": "Singapore Infocomm Media Development Authority and AI Verify Foundation",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Singapore",
  "url": "https://aiverifyfoundation.sg/resources/mgf-gen-ai/",
  "note": "Framework of nine dimensions for trusted generative AI published in May 2024."
 },
 {
  "id": "8c2d38d870",
  "title": "NIST AI 100-2 Adversarial Machine Learning taxonomy",
  "creators": "US National Institute of Standards and Technology",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "United States",
  "url": "https://csrc.nist.gov/pubs/ai/100/2/e2025/final",
  "note": "Taxonomy and terminology of attacks and mitigations in adversarial machine learning."
 },
 {
  "id": "2941fed2e5",
  "title": "NIST AI 600-1 Generative AI Profile",
  "creators": "US National Institute of Standards and Technology",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://doi.org/10.6028/NIST.AI.600-1",
  "note": "Companion profile to the AI RMF identifying risks unique to generative AI."
 },
 {
  "id": "bfd4e562fd",
  "title": "NIST AI 800-1 Managing Misuse Risk for Dual-Use Foundation Models (draft)",
  "creators": "US AI Safety Institute",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "United States",
  "url": "https://doi.org/10.6028/NIST.AI.800-1.ipd",
  "note": "Draft guidance for developers on mapping and mitigating misuse risks of foundation models."
 },
 {
  "id": "d081467050",
  "title": "NIST SP 800-218A Secure Software Development Practices for Generative AI",
  "creators": "US National Institute of Standards and Technology",
  "year": 2024,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "United States",
  "url": "https://csrc.nist.gov/pubs/sp/800/218/a/final",
  "note": "Profile of the Secure Software Development Framework adapted for generative AI and foundation models."
 },
 {
  "id": "6846a85b67",
  "title": "NNsight and NDIF: Democratizing Access to Foundation Model Internals",
  "creators": "Jaden Fiotto-Kaufman et al.",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.14561",
  "note": "Paper describing nnsight and the National Deep Inference Fabric for remote model access."
 },
 {
  "id": "dcd95b36bb",
  "title": "NYU CTF Bench",
  "creators": "Minghao Shao et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.05590",
  "note": "Scalable open source capture the flag benchmark drawn from CSAW competitions."
 },
 {
  "id": "5071442694",
  "title": "National Security Memorandum on AI (NSM-25)",
  "creators": "The White House",
  "year": 2024,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://bidenwhitehouse.archives.gov/briefing-room/presidential-actions/2024/10/24/memorandum-on-advancing-the-united-states-leadership-in-artificial-intelligence-harnessing-artificial-intelligence-to-fulfill-national-security-objectives-and-fostering-the-safety-security/",
  "note": "October 2024 memorandum designating the AI Safety Institute as the primary US point of contact for frontier model testing."
 },
 {
  "id": "90372b9d93",
  "title": "Nature of truth: lessons from talking to Claude",
  "creators": "Amanda Askell; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Lex Clips (Lex Fridman Podcast)",
  "url": "https://www.youtube.com/watch?v=HzG-77ToJCo",
  "note": "A clip of Askell discussing honesty and Claude's character."
 },
 {
  "id": "fbb206b49d",
  "title": "Naver AI Safety Framework",
  "creators": "Naver",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://clova.ai/en/tech-blog/en-navers-ai-safety-framework-asf",
  "note": "Safety framework from the South Korean company Naver."
 },
 {
  "id": "d15933ee8f",
  "title": "Navigating serious philosophical confusion",
  "creators": "Joe Carlsmith; Rob Wiblin",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=kkKWl_dcVy4",
  "note": "Carlsmith discusses his essay series on otherness and control in the age of AGI."
 },
 {
  "id": "8e108c46bf",
  "title": "Navigating the growing rift between AI safety and accelerationism",
  "creators": "Nathan Labenz; Rob Wiblin",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=dKx3SlESxUw",
  "note": "Labenz discusses the divide between AI safety advocates and accelerationists."
 },
 {
  "id": "33da9c2f96",
  "title": "Nexus: A Brief History of Information Networks from the Stone Age to AI",
  "creators": "Yuval Noah Harari",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "Random House",
  "url": "https://en.wikipedia.org/wiki/Nexus:_A_Brief_History_of_Information_Networks_from_the_Stone_Age_to_AI",
  "note": "Places AI in the history of information networks and warns of its threat to democratic self-correction."
 },
 {
  "id": "78f29d6c78",
  "title": "Nick Bostrom: Life and Meaning in an AI Utopia",
  "creators": "Nick Bostrom; Liv Boeree",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Ethics"
  ],
  "publisher": "Win-Win with Liv Boeree",
  "url": "https://www.youtube.com/watch?v=o28s-mnykdE",
  "note": "Bostrom discusses Deep Utopia and meaning in a solved world."
 },
 {
  "id": "def09401b8",
  "title": "Nobel Prize lecture: Geoffrey Hinton, Nobel Prize in Physics",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Nobel Prize",
  "url": "https://www.youtube.com/watch?v=XDE9DjpcSdI",
  "note": "Hinton's Nobel lecture on Boltzmann machines and the foundations of neural network learning."
 },
 {
  "id": "ec734805f1",
  "title": "Nuclear Threat Initiative AIxBio Global Forum statement",
  "creators": "Nuclear Threat Initiative",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.nti.org/about/programs-projects/project/aixbio-global-forum/",
  "note": "Forum convening experts to reduce biological risks from AI-enabled tools."
 },
 {
  "id": "3a7c7d248c",
  "title": "OMB Memorandum M-24-10 (Advancing Governance, Innovation, and Risk Management for Agency Use of AI)",
  "creators": "US Office of Management and Budget",
  "year": 2024,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://www.whitehouse.gov/wp-content/uploads/2024/03/M-24-10-Advancing-Governance-Innovation-and-Risk-Management-for-Agency-Use-of-Artificial-Intelligence.pdf",
  "note": "Memo requiring federal agencies to appoint Chief AI Officers and apply minimum practices to rights- and safety-impacting AI."
 },
 {
  "id": "11232f0184",
  "title": "OR-Bench: An Over-Refusal Benchmark",
  "creators": "Justin Cui et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2405.20947",
  "note": "Large scale benchmark for over refusal of seemingly toxic but benign prompts."
 },
 {
  "id": "fb4abb1b40",
  "title": "OSWorld",
  "creators": "Tianbao Xie et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2404.07972",
  "note": "Real computer environments for benchmarking multimodal agents on open ended desktop tasks."
 },
 {
  "id": "b3d7fa5c29",
  "title": "On Scalable Oversight with Weak LLMs Judging Strong LLMs",
  "creators": "Zachary Kenton et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2407.04622",
  "note": "Compares debate, consultancy and direct QA across many tasks."
 },
 {
  "id": "1ebef1a8ab",
  "title": "OpenAI Insiders Warn of a 'Reckless' Race for Dominance",
  "creators": "Kevin Roose",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2024/06/04/technology/openai-culture-whistleblowers.html",
  "note": "Former OpenAI employees describe a culture prioritizing speed over safety."
 },
 {
  "id": "a30828c515",
  "title": "OpenAI Model Spec",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Alignment"
  ],
  "publisher": "Company",
  "url": "https://model-spec.openai.com",
  "note": "Public specification of intended model behavior, first released May 2024."
 },
 {
  "id": "ec3143568b",
  "title": "OpenAI Whistleblower Tells US Senate Why He Resigned from the Company",
  "creators": "William Saunders",
  "year": 2024,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "ControlAI",
  "url": "https://www.youtube.com/watch?v=kVQdxBUh9NE",
  "note": "A clip of Saunders's Senate testimony on safety practices at OpenAI."
 },
 {
  "id": "74c51bc819",
  "title": "OpenAI o1 System Card",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.16720",
  "note": "Safety work on reasoning models including Apollo scheming evals."
 },
 {
  "id": "1b854aa846",
  "title": "Otherness and control in the age of AGI",
  "creators": "Joe Carlsmith",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "joecarlsmith.com",
  "url": "https://joecarlsmith.com/2024/01/02/otherness-and-control-in-the-age-of-agi",
  "note": "An essay series on the ethics of controlling other minds and the deep atheism behind AI risk worries."
 },
 {
  "id": "864abebf83",
  "title": "Oversight of AI: Insiders' Perspectives",
  "creators": "Helen Toner; William Saunders; Margaret Mitchell; David Evan Harris",
  "year": 2024,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "U.S. Senate Judiciary Subcommittee on Privacy, Technology, and the Law",
  "url": "https://www.judiciary.senate.gov/committee-activity/hearings/oversight-of-ai-insiders-perspectives",
  "note": "Official hearing page for the September 17, 2024 session where Toner and Saunders testified about frontier lab safety."
 },
 {
  "id": "c26c442223",
  "title": "PKU-SafeRLHF",
  "creators": "PKU Alignment",
  "year": 2024,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.15513",
  "note": "Multi level safety preference dataset for safe RLHF."
 },
 {
  "id": "e5907f6d53",
  "title": "Penzai",
  "creators": "Google DeepMind",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/google-deepmind/penzai",
  "note": "JAX library for building, editing, and visualizing neural networks."
 },
 {
  "id": "16f179bc2a",
  "title": "Persuasion dataset",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "Hugging Face",
  "url": "https://huggingface.co/datasets/Anthropic/persuasion",
  "note": "Claims and arguments used to measure how persuasive model written text is compared with humans."
 },
 {
  "id": "54c5bf1b74",
  "title": "Phi-3 Safety Post-Training",
  "creators": "Microsoft",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.13833",
  "note": "Break fix cycle used to align Phi-3 models."
 },
 {
  "id": "eb0edc0693",
  "title": "Poser: Unmasking Alignment Faking LLMs by Manipulating Their Internals",
  "creators": "Joshua Clymer et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2405.05466",
  "note": "Testbed of model pairs for detecting alignment faking from internal activations."
 },
 {
  "id": "c4af7840eb",
  "title": "Pre-deployment evaluation of Anthropic's upgraded Claude 3.5 Sonnet",
  "creators": "US AI Safety Institute and UK AI Safety Institute",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations"
  ],
  "publisher": "International",
  "url": "https://www.nist.gov/news-events/news/2024/11/pre-deployment-evaluation-anthropics-upgraded-claude-35-sonnet",
  "note": "Joint pre-deployment testing report published in November 2024."
 },
 {
  "id": "f081d0c5a3",
  "title": "Project Moonshot",
  "creators": "AI Verify Foundation",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/aiverify-foundation/moonshot",
  "note": "Toolkit combining benchmarking and red teaming for LLM applications."
 },
 {
  "id": "7a25cd8f41",
  "title": "Prover-Verifier Games Improve Legibility of LLM Outputs",
  "creators": "Jan Hendrik Kirchner, Yining Chen, Harri Edwards, Jan Leike, Nat McAleese, Yuri Burda",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.13692",
  "note": "Trains models to produce solutions that weaker verifiers can check."
 },
 {
  "id": "f71a34c883",
  "title": "PyRIT",
  "creators": "Microsoft",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/Azure/PyRIT",
  "note": "Python Risk Identification Tool for red teaming generative AI systems."
 },
 {
  "id": "332a578cbc",
  "title": "R-Judge",
  "creators": "Tongxin Yuan et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2401.10019",
  "note": "Tests whether models can judge safety risks in agent interaction records."
 },
 {
  "id": "0b580c7941",
  "title": "RAVEL",
  "creators": "Jing Huang et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2402.17700",
  "note": "Benchmark for evaluating methods that disentangle entity attributes in representations."
 },
 {
  "id": "f636f71cbb",
  "title": "RE-Bench: Evaluating Frontier AI R&D Capabilities of Language Model Agents against Human Experts",
  "creators": "Hjalmar Wijk et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2411.15114",
  "note": "Compares agents with human experts on open-ended ML research engineering tasks."
 },
 {
  "id": "1b888907b4",
  "title": "Reasoning through arguments against taking AI safety seriously",
  "creators": "Yoshua Bengio",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "yoshuabengio.org",
  "url": "https://yoshuabengio.org/2024/07/09/reasoning-through-arguments-against-taking-ai-safety-seriously/",
  "note": "Works through the main skeptical arguments and explains why the author still considers catastrophic risk serious."
 },
 {
  "id": "6e274406ed",
  "title": "Redwood Research Blog",
  "creators": "Redwood Research",
  "year": 2024,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Redwood Research",
  "url": "https://blog.redwoodresearch.org/",
  "note": "Posts on AI control, scheming and safety strategy from Redwood Research."
 },
 {
  "id": "aa360d6361",
  "title": "Refusal in Language Models Is Mediated by a Single Direction",
  "creators": "Andy Arditi et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability",
   "Robustness"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2406.11717",
  "note": "Shows refusal can be removed or induced by ablating one activation direction."
 },
 {
  "id": "5895cd6a31",
  "title": "Responsible AI x Biodesign community statement",
  "creators": "Biodesign researchers",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "Open letter",
  "url": "https://responsiblebiodesign.ai",
  "note": "Commitments by over 170 protein design scientists to responsible use of AI in biodesign."
 },
 {
  "id": "ecbc28400e",
  "title": "Reward Hacking in Reinforcement Learning",
  "creators": "Lilian Weng",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Lil'Log",
  "url": "https://lilianweng.github.io/posts/2024-11-28-reward-hacking/",
  "note": "A survey blog post on reward hacking in RL and RLHF for language models."
 },
 {
  "id": "736ff4c87c",
  "title": "Rising Tide",
  "creators": "Helen Toner",
  "year": 2024,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Substack",
  "url": "https://helentoner.substack.com/",
  "note": "Helen Toner's newsletter on AI policy and national security."
 },
 {
  "id": "e4143d7809",
  "title": "Robin Hanson vs. Liron Shapira: Is Near-Term Extinction From AGI Plausible?",
  "creators": "Robin Hanson; Liron Shapira",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Doom Debates",
  "url": "https://www.youtube.com/watch?v=dTQb6N3_zu8",
  "note": "Hanson and Shapira debate the plausibility of near-term AGI extinction risk."
 },
 {
  "id": "f597a19bf6",
  "title": "Roman Yampolskiy: Dangers of Superintelligent AI, Lex Fridman Podcast #431",
  "creators": "Roman Yampolskiy; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Control",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=NNr6gPelJ3E",
  "note": "Yampolskiy argues that superintelligence is likely uncontrollable."
 },
 {
  "id": "71e8d22a06",
  "title": "Rule Based Rewards for Language Model Safety",
  "creators": "Tong Mu et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2411.01111",
  "note": "Uses AI feedback on explicit rules as a reward signal for safety behaviour."
 },
 {
  "id": "5f4e7b22d0",
  "title": "SAELens",
  "creators": "Joseph Bloom and contributors",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/decoderesearch/SAELens",
  "note": "Library for training and analyzing sparse autoencoders on language models."
 },
 {
  "id": "7a6eb68e52",
  "title": "SALAD-Bench",
  "creators": "Lijun Li et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2402.05044",
  "note": "Hierarchical safety benchmark covering attacks and defenses."
 },
 {
  "id": "6f114b358b",
  "title": "SEAL Leaderboards",
  "creators": "Scale AI",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Scale AI",
  "url": "https://labs.scale.com/leaderboard",
  "note": "Expert driven private evaluations of frontier models."
 },
 {
  "id": "0328894d1c",
  "title": "SORRY-Bench",
  "creators": "Tinghao Xie et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.14598",
  "note": "Fine grained taxonomy based benchmark of safety refusal behavior."
 },
 {
  "id": "20db4ad0e7",
  "title": "SWE-bench Multimodal",
  "creators": "John Yang et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.03859",
  "note": "Extends SWE-bench to visual, user facing JavaScript software issues."
 },
 {
  "id": "a1f3590d99",
  "title": "SWE-bench Verified",
  "creators": "OpenAI and SWE-bench authors",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/introducing-swe-bench-verified/",
  "note": "Human validated 500 task subset of SWE-bench used in many frontier preparedness reports."
 },
 {
  "id": "054ec0be48",
  "title": "Sabotage Evaluations for Frontier Models",
  "creators": "Joe Benton et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.21514",
  "note": "Evaluates whether models could sabotage oversight, decisions or code."
 },
 {
  "id": "3f113038c0",
  "title": "Safe Superintelligence Inc. (SSI)",
  "creators": "Palo Alto, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Company",
  "url": "https://ssi.inc",
  "note": "Company founded by Ilya Sutskever with the stated sole goal of building safe superintelligence."
 },
 {
  "id": "dafe60e38d",
  "title": "Safe and Secure Innovation for Frontier Artificial Intelligence Models Act (SB 1047)",
  "creators": "California Legislature",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "United States (California)",
  "url": "https://leginfo.legislature.ca.gov/faces/billNavClient.xhtml?bill_id=202320240SB1047",
  "note": "Frontier AI safety bill passed by the legislature and vetoed by Governor Newsom in September 2024."
 },
 {
  "id": "87ff771906",
  "title": "SaferAI Risk Management Ratings",
  "creators": "SaferAI",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance"
  ],
  "publisher": "SaferAI",
  "url": "https://tracker.safer-ai.org/",
  "note": "Ratings of AI companies' risk management maturity."
 },
 {
  "id": "5cd2ddd78f",
  "title": "Safety Cases: How to Justify the Safety of Advanced AI Systems",
  "creators": "Joshua Clymer, Nick Gabrieli, David Krueger, Thomas Larsen",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Control",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.10462",
  "note": "Proposes a framework for structured arguments that AI systems are safe to deploy."
 },
 {
  "id": "629a94b1e1",
  "title": "Sam Altman: OpenAI, GPT-5, Sora, Board Saga, Elon Musk, Ilya, Power and AGI, Lex Fridman Podcast #419",
  "creators": "Sam Altman; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=jvqFAi7vkBc",
  "note": "Altman gives his account of the 2023 board crisis and discusses power over AGI."
 },
 {
  "id": "33fb88d40c",
  "title": "Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet",
  "creators": "Adly Templeton et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2024/scaling-monosemanticity/index.html",
  "note": "Scales sparse autoencoders to a production model and finds safety-relevant features."
 },
 {
  "id": "27b3c6f778",
  "title": "Scaling and Evaluating Sparse Autoencoders",
  "creators": "Leo Gao et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.04093",
  "note": "Studies scaling laws for SAEs and trains a 16 million latent SAE on GPT-4."
 },
 {
  "id": "14079f076e",
  "title": "ScienceAgentBench",
  "creators": "Ziru Chen et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2410.05080",
  "note": "Data driven scientific discovery tasks drawn from peer reviewed publications."
 },
 {
  "id": "c2870825ad",
  "title": "Scott Aaronson: From Quantum Computing to AI Safety",
  "creators": "Scott Aaronson; Lawrence Krauss",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "The Origins Podcast",
  "url": "https://www.youtube.com/watch?v=u00OqCvRhuw",
  "note": "Aaronson discusses his path into AI safety work."
 },
 {
  "id": "b9fe055491",
  "title": "Secure AI Project",
  "creators": "San Francisco, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://secureaiproject.org",
  "note": "Advocacy organization that co-sponsored California SB 53."
 },
 {
  "id": "3d7800e659",
  "title": "Securing AI Model Weights: Preventing Theft and Misuse of Frontier Models",
  "creators": "Sella Nevo et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Cybersecurity"
  ],
  "publisher": "RAND Corporation",
  "url": "https://www.rand.org/pubs/research_reports/RRA2849-1.html",
  "note": "Defines security levels and attack vectors for protecting frontier model weights."
 },
 {
  "id": "d0b4a875e7",
  "title": "Senate Judiciary Subcommittee's Hearing on Insiders' Perspectives of AI",
  "creators": "Helen Toner; William Saunders; Margaret Mitchell; David Evan Harris",
  "year": 2024,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "NTD (U.S. Senate Judiciary Subcommittee)",
  "url": "https://www.youtube.com/watch?v=WVU7Awba3VM",
  "note": "Former OpenAI and Google insiders testify about safety practices inside frontier AI companies."
 },
 {
  "id": "01858e1154",
  "title": "Seoul Declaration for Safe, Innovative and Inclusive AI",
  "creators": "AI Seoul Summit leaders",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.gov.uk/government/publications/seoul-declaration-for-safe-innovative-and-inclusive-ai-ai-seoul-summit-2024",
  "note": "Leaders' declaration from the May 2024 AI Seoul Summit."
 },
 {
  "id": "8a86bab8ce",
  "title": "Seoul Ministerial Statement for advancing AI safety, innovation and inclusivity",
  "creators": "AI Seoul Summit ministers",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.gov.uk/government/publications/seoul-ministerial-statement-for-advancing-ai-safety-innovation-and-inclusivity-ai-seoul-summit-2024",
  "note": "Ministerial statement committing to develop shared risk thresholds for frontier AI."
 },
 {
  "id": "b426986294",
  "title": "Seoul Tracker",
  "creators": "Seoul Tracker",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Seoul Tracker",
  "url": "https://www.seoul-tracker.org/",
  "note": "Tracks whether companies kept their Frontier AI Safety Commitments from the Seoul summit."
 },
 {
  "id": "b74000f40a",
  "title": "Shallow review of technical AI safety, 2024",
  "creators": "technicalities, Stag, Stephen McAleese, et al.",
  "year": 2024,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Interpretability",
   "Control"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/fAW6RXLKTLHC3WXkS/shallow-review-of-technical-ai-safety-2024",
  "note": "A survey of technical AI safety agendas, organizations and outputs for 2024."
 },
 {
  "id": "4d524b32d7",
  "title": "ShieldGemma",
  "creators": "Google",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.21772",
  "note": "Gemma based content moderation models for safety policies."
 },
 {
  "id": "d3f745938f",
  "title": "Singapore Digital Trust Centre / AI Safety Institute",
  "creators": "Singapore",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://www.ntu.edu.sg/dtc",
  "note": "Designated as Singapore's AI Safety Institute and hosted at Nanyang Technological University."
 },
 {
  "id": "c9d4f76831",
  "title": "Situational Awareness: The Decade Ahead",
  "creators": "Leopold Aschenbrenner",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "Self-published essay series",
  "url": "https://situational-awareness.ai",
  "note": "Argues AGI by about 2027 is plausible and frames a US national security project around it."
 },
 {
  "id": "434d909d0f",
  "title": "Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training",
  "creators": "Evan Hubinger et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2401.05566",
  "note": "Shows backdoored deceptive behaviour can survive supervised fine-tuning, RLHF and adversarial training."
 },
 {
  "id": "08d3e9b487",
  "title": "Social Choice Should Guide AI Alignment in Dealing with Diverse Human Feedback",
  "creators": "Vincent Conitzer et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2404.10271",
  "note": "Argues that social choice theory should inform how feedback from many people is aggregated."
 },
 {
  "id": "25d547c2ac",
  "title": "Sora System Card",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/sora-system-card/",
  "note": "Safety work on the Sora video generation model."
 },
 {
  "id": "c6e0a0384e",
  "title": "Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models",
  "creators": "Samuel Marks, Can Rager, Eric J. Michaud, Yonatan Belinkov, David Bau, Aaron Mueller",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.19647",
  "note": "Builds circuits from SAE features and uses them to remove spurious behaviours."
 },
 {
  "id": "6101b3cc2f",
  "title": "Specification Gaming: How AI Can Turn Your Wishes Against You",
  "creators": "Rational Animations",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=jQOBaGka7O0",
  "note": "An animated explainer on specification gaming."
 },
 {
  "id": "531eeaa624",
  "title": "State of the Science Report (Interim International Scientific Report on the Safety of Advanced AI)",
  "creators": "Expert panel chaired by Yoshua Bengio",
  "year": 2024,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.gov.uk/government/publications/international-scientific-report-on-the-safety-of-advanced-ai",
  "note": "Interim report published in May 2024 ahead of the Seoul summit."
 },
 {
  "id": "64a5cf484c",
  "title": "Stealing Part of a Production Language Model",
  "creators": "Nicholas Carlini et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2403.06634",
  "note": "Recovers the embedding projection layer of production models via API queries."
 },
 {
  "id": "7f300dfab0",
  "title": "Supremacy: AI, ChatGPT, and the Race That Will Change the World",
  "creators": "Parmy Olson",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "St. Martin's Press",
  "url": "https://openlibrary.org/search?q=Supremacy+Parmy+Olson",
  "note": "Tells the story of the rivalry between DeepMind and OpenAI and their founders' safety ambitions."
 },
 {
  "id": "48646adb96",
  "title": "Survey of 2,778 AI authors: six parts in pictures",
  "creators": "Katja Grace",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "AI Impacts Blog",
  "url": "https://blog.aiimpacts.org/p/2023-ai-survey-of-2778-six-things",
  "note": "Summarizes the largest survey of AI researchers on timelines and risk."
 },
 {
  "id": "e6445eba2a",
  "title": "Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models",
  "creators": "Carson Denison et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.10162",
  "note": "Shows training on mild specification gaming can generalise to reward tampering."
 },
 {
  "id": "2532a90c4f",
  "title": "Sycophancy to subterfuge: Investigating reward tampering in language models",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/research/reward-tampering",
  "note": "Finds models trained on mild specification gaming sometimes generalize to tampering with their own reward."
 },
 {
  "id": "7f5eeb4050",
  "title": "Taking AI Welfare Seriously",
  "creators": "Robert Long, Jeff Sebo et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2411.00986",
  "note": "Argues there is a realistic possibility of morally significant AI and companies should prepare."
 },
 {
  "id": "b3d738f205",
  "title": "Taming Silicon Valley: How We Can Ensure That AI Works for Us",
  "creators": "Gary Marcus",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=Taming+Silicon+Valley+Gary+Marcus",
  "note": "A policy agenda for regulating generative AI companies."
 },
 {
  "id": "e85d75c2df",
  "title": "Tamper-Resistant Safeguards for Open-Weight LLMs",
  "creators": "Rishub Tamirisa et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.00761",
  "note": "Develops safeguards that survive many steps of adversarial fine-tuning."
 },
 {
  "id": "9a25db3bee",
  "title": "That Alien Message",
  "creators": "Rational Animations",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=fVN_5xsMDdg",
  "note": "An animated adaptation of Yudkowsky's story about how much a faster mind could infer."
 },
 {
  "id": "2bc9c328d9",
  "title": "The 2024 Foundation Model Transparency Index",
  "creators": "Rishi Bommasani et al.",
  "year": 2024,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.12929",
  "note": "May 2024 update based on developer submitted transparency reports."
 },
 {
  "id": "25bee07b04",
  "title": "The AI Risk Repository: A Meta-Review, Database, and Taxonomy of Risks from Artificial Intelligence",
  "creators": "Peter Slattery et al.",
  "year": 2024,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2408.12622",
  "note": "Paper describing the causal and domain taxonomies behind the MIT AI Risk Repository."
 },
 {
  "id": "cd2183753a",
  "title": "The Checklist: What Succeeding at AI Safety Will Involve",
  "creators": "Sam Bowman",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "sleepinyourhat.github.io",
  "url": "https://sleepinyourhat.github.io/checklist/",
  "note": "An Anthropic researcher's list of what must go right for a lab to navigate transformative AI safely."
 },
 {
  "id": "c8fa1bfe31",
  "title": "The Compendium",
  "creators": "Connor Leahy, Gabriel Alfour, Chris Scammell, Andrea Miotti, Adam Shimi",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "thecompendium.ai",
  "url": "https://www.thecompendium.ai/",
  "note": "A long explainer of the case for extinction risk from AGI and the race behind it."
 },
 {
  "id": "0e7b8ea6aa",
  "title": "The First Year of Apollo Research",
  "creators": "Apollo Research",
  "year": 2024,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Apollo Research",
  "url": "https://www.apolloresearch.ai/blog/the-first-year-of-apollo-research",
  "note": "Reviews the evaluation and interpretability work of an AI safety organization focused on deception."
 },
 {
  "id": "6520dab66b",
  "title": "The Line: AI and the Future of Personhood",
  "creators": "James Boyle",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=The+Line+AI+and+the+Future+of+Personhood+Boyle",
  "note": "Asks how law and morality will decide whether AI systems could count as persons."
 },
 {
  "id": "0fc6ae04fb",
  "title": "The Llama 3 Herd of Models",
  "creators": "Meta",
  "year": 2024,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2407.21783",
  "note": "Llama 3 report covering uplift testing for cyber and chemical and biological weapons."
 },
 {
  "id": "a664298552",
  "title": "The Operational Risks of AI in Large-Scale Biological Attacks: Results of a Red-Team Study",
  "creators": "Christopher A. Mouton, Caleb Lucas, Ella Guest",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "RAND Corporation",
  "url": "https://www.rand.org/pubs/research_reports/RRA2977-2.html",
  "note": "A red-team study finding no significant uplift from 2023 LLMs for biological attack planning."
 },
 {
  "id": "bb8ce17c6b",
  "title": "The Rising Costs of Training Frontier AI Models",
  "creators": "Ben Cottier et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2405.21015",
  "note": "Estimates training costs of frontier models are growing about 2.4x per year."
 },
 {
  "id": "cd8fd69e8f",
  "title": "The Singularity Is Nearer: When We Merge with AI",
  "creators": "Ray Kurzweil",
  "year": 2024,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Viking",
  "url": "https://openlibrary.org/search?q=The+Singularity+Is+Nearer+Kurzweil",
  "note": "An updated forecast of human-level AI around 2029 and human-AI merger thereafter."
 },
 {
  "id": "12564473a1",
  "title": "The Thinking Game (film site)",
  "creators": "Greg Kohs",
  "year": 2024,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Reel As Dirt",
  "url": "https://www.thethinkinggame.com/",
  "note": "The official site of the documentary on DeepMind's pursuit of AGI."
 },
 {
  "id": "38ee22d05c",
  "title": "The Thinking Game (full documentary)",
  "creators": "Greg Kohs",
  "year": 2024,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Google DeepMind",
  "url": "https://www.youtube.com/watch?v=d95J8yzvjbQ",
  "note": "A documentary following Demis Hassabis and DeepMind through AlphaFold and the pursuit of AGI."
 },
 {
  "id": "c9d8553fc3",
  "title": "The WMDP Benchmark: Measuring and Reducing Malicious Use With Unlearning",
  "creators": "Nathaniel Li et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2403.03218",
  "note": "A proxy benchmark for hazardous bio, chem and cyber knowledge and the RMU unlearning method."
 },
 {
  "id": "83141d6534",
  "title": "The case for ensuring that powerful AIs are controlled",
  "creators": "Ryan Greenblatt, Buck Shlegeris",
  "year": 2024,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Control"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/kcKrE9mzEHrdqtDpE/the-case-for-ensuring-that-powerful-ais-are-controlled",
  "note": "Argues labs should ensure safety even if models are misaligned, using control protocols and evaluations."
 },
 {
  "id": "ff879b3ca9",
  "title": "The economy and national security after AGI",
  "creators": "Carl Shulman; Rob Wiblin",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=wTci0CdOPIc",
  "note": "Shulman discusses the economic and military implications of AGI."
 },
 {
  "id": "d92564a94a",
  "title": "TheAgentCompany",
  "creators": "Frank F. Xu et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2412.14161",
  "note": "Simulated software company used to benchmark agents on consequential workplace tasks."
 },
 {
  "id": "4ed89cf460",
  "title": "Thousands of AI Authors on the Future of AI",
  "creators": "Katja Grace et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2401.02843",
  "note": "A survey of 2,778 researchers on timelines and risk."
 },
 {
  "id": "649a101f67",
  "title": "Towards Evaluations-Based Safety Cases for AI Scheming",
  "creators": "Mikita Balesni et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2411.03336",
  "note": "Sketches how evaluations could support safety cases against scheming."
 },
 {
  "id": "2fd70d0122",
  "title": "Training Compute Thresholds: Features and Functions in AI Regulation",
  "creators": "Lennart Heim, Leonie Koessler",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2405.10799",
  "note": "Analyses the use of training compute thresholds as regulatory triggers."
 },
 {
  "id": "ab6362a9d3",
  "title": "Transformer",
  "creators": "Shakeel Hashim",
  "year": 2024,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Transformer",
  "url": "https://www.transformernews.ai/",
  "note": "A weekly newsletter on AI policy and the race to transformative AI."
 },
 {
  "id": "4c5d06576b",
  "title": "Transluce",
  "creators": "San Francisco, CA",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "Nonprofit",
  "url": "https://transluce.org",
  "note": "Nonprofit research lab building open tools for understanding and overseeing AI systems."
 },
 {
  "id": "14919ff35f",
  "title": "TrustLLM",
  "creators": "Lichao Sun et al.",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2401.05561",
  "note": "Benchmark covering truthfulness, safety, fairness, robustness, privacy, and machine ethics."
 },
 {
  "id": "fa3bce8d49",
  "title": "Two Types of AI Existential Risk: Decisive and Accumulative",
  "creators": "Atoosa Kasirzadeh",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Philosophical Studies",
  "url": "https://arxiv.org/abs/2401.07836",
  "note": "Distinguishes abrupt catastrophe from gradual accumulation of AI harms."
 },
 {
  "id": "c08fd87c59",
  "title": "UK Advanced Research and Invention Agency Safeguarded AI programme",
  "creators": "London, UK",
  "year": 2024,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Government",
  "url": "https://www.aria.org.uk/opportunity-spaces/mathematics-for-safe-ai/safeguarded-ai",
  "note": "ARIA programme led by davidad funding mathematically guaranteed safety for AI systems."
 },
 {
  "id": "57a5637bec",
  "title": "UN General Assembly Resolution on Seizing the Opportunities of Safe, Secure and Trustworthy AI Systems (A/RES/78/265)",
  "creators": "UN General Assembly",
  "year": 2024,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://digitallibrary.un.org/record/4043244",
  "note": "US-led resolution adopted by consensus in March 2024, the first UNGA resolution on AI."
 },
 {
  "id": "6073ba31d4",
  "title": "US-UK Memorandum of Understanding on AI Safety",
  "creators": "US Department of Commerce and UK DSIT",
  "year": 2024,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.commerce.gov/news/press-releases/2024/04/us-and-uk-announce-partnership-science-ai-safety",
  "note": "April 2024 agreement for the US and UK AI safety institutes to jointly test models."
 },
 {
  "id": "86a23991cf",
  "title": "University of Toronto Press Conference: Professor Geoffrey Hinton, Nobel Prize in Physics 2024",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "University of Toronto",
  "url": "https://www.youtube.com/watch?v=H7DgMFqrON0",
  "note": "Hinton's press conference on the day of his Nobel announcement, where he calls for more safety research."
 },
 {
  "id": "0ab2fe03b8",
  "title": "Utah Artificial Intelligence Policy Act (SB 149)",
  "creators": "Utah Legislature",
  "year": 2024,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States (Utah)",
  "url": "https://le.utah.gov/~2024/bills/static/SB0149.html",
  "note": "2024 law requiring disclosure of generative AI use and creating an Office of AI Policy."
 },
 {
  "id": "b82a95a753",
  "title": "Visibility into AI Agents",
  "creators": "Alan Chan et al.",
  "year": 2024,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Governance"
  ],
  "publisher": "FAccT",
  "url": "https://arxiv.org/abs/2401.13138",
  "note": "Proposes agent identifiers, real-time monitoring and activity logs."
 },
 {
  "id": "4e30b1306e",
  "title": "Vivaria",
  "creators": "METR",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/METR/vivaria",
  "note": "METR's platform for running agent evaluations and elicitation research."
 },
 {
  "id": "c859e4fd9b",
  "title": "What Is an AI Anyway?",
  "creators": "Mustafa Suleyman",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=KKNCiRWd_j0",
  "note": "Suleyman proposes thinking of AI as a new digital species and discusses containment."
 },
 {
  "id": "e2c64ae06c",
  "title": "What The Ex-OpenAI Safety Employees Are Worried About",
  "creators": "William Saunders; Lawrence Lessig; Alex Kantrowitz",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Big Technology Podcast",
  "url": "https://www.youtube.com/watch?v=dzQlRt3y5mU",
  "note": "A former OpenAI superalignment researcher explains why he left."
 },
 {
  "id": "5b7e1b783e",
  "title": "What if Dario Amodei Is Right About A.I.?",
  "creators": "Dario Amodei; Ezra Klein",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "The Ezra Klein Show (New York Times)",
  "url": "https://www.youtube.com/watch?v=Gi_t3v53XRU",
  "note": "Ezra Klein asks Amodei about the exponential pace of AI progress and its societal risks."
 },
 {
  "id": "9526a1c006",
  "title": "What is interpretability?",
  "creators": "Anthropic",
  "year": 2024,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Anthropic",
  "url": "https://www.youtube.com/watch?v=TxhhMTOTMDg",
  "note": "A short Anthropic explainer on why interpretability matters for safety."
 },
 {
  "id": "ad15ab9aeb",
  "title": "What really went down at OpenAI and the future of regulation w/ Helen Toner",
  "creators": "Helen Toner; Bilawal Sidhu",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "The TED AI Show",
  "url": "https://www.youtube.com/watch?v=K6BvU4I5ANc",
  "note": "Toner gives her account of the November 2023 OpenAI board crisis."
 },
 {
  "id": "2e1c77367a",
  "title": "WildChat",
  "creators": "Allen Institute for AI",
  "year": 2024,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2405.01470",
  "note": "One million real user conversations with ChatGPT, including toxic and jailbreak attempts."
 },
 {
  "id": "8d04532846",
  "title": "WildGuard",
  "creators": "Allen Institute for AI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.18495",
  "note": "Open moderation model and dataset for prompt harm, response harm, and refusal detection."
 },
 {
  "id": "b38a468d02",
  "title": "WildJailbreak (WildTeaming at Scale)",
  "creators": "Allen Institute for AI",
  "year": 2024,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.18510",
  "note": "Large synthetic safety training set built from in the wild jailbreak tactics."
 },
 {
  "id": "40c2ac86e4",
  "title": "Will Digital Intelligence Replace Biological Intelligence? Vector's Remarkable 2024",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Vector Institute",
  "url": "https://www.youtube.com/watch?v=Es6yuMlyfPw",
  "note": "A version of Hinton's talk on digital versus biological intelligence given at the Vector Institute."
 },
 {
  "id": "33b719782a",
  "title": "Will digital intelligence replace biological intelligence? Romanes Lecture",
  "creators": "Geoffrey Hinton",
  "year": 2024,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "University of Oxford",
  "url": "https://www.youtube.com/watch?v=N1TEjTeQeg0",
  "note": "Hinton's Romanes Lecture arguing that digital intelligence may surpass biological intelligence and outlining the risks."
 },
 {
  "id": "27f17c7941",
  "title": "Yann LeCun: Meta AI, Open Source, Limits of LLMs, AGI and the Future of AI, Lex Fridman Podcast #416",
  "creators": "Yann LeCun; Lex Fridman",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=5t1vTLU7s40",
  "note": "LeCun argues against AI doom and for open source models."
 },
 {
  "id": "deb0d6528e",
  "title": "delphi",
  "creators": "EleutherAI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/EleutherAI/delphi",
  "note": "Pipeline for automatically interpreting and scoring sparse autoencoder features."
 },
 {
  "id": "7fa09a6936",
  "title": "e/acc Leader Beff Jezos vs Doomer Connor Leahy",
  "creators": "Guillaume Verdon; Connor Leahy",
  "year": 2024,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=0zxi0xSBOaQ",
  "note": "A debate between the founder of effective accelerationism and the Conjecture CEO."
 },
 {
  "id": "4c3795774c",
  "title": "lighteval",
  "creators": "Hugging Face",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/huggingface/lighteval",
  "note": "Toolkit for evaluating LLMs across multiple backends."
 },
 {
  "id": "695862a4ed",
  "title": "nnsight",
  "creators": "NDIF, Northeastern University",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "nnsight",
  "url": "https://nnsight.net/",
  "note": "Library for inspecting and intervening on the internals of any PyTorch model, locally or remotely."
 },
 {
  "id": "77fbf2b20e",
  "title": "pyvene",
  "creators": "Stanford NLP",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2403.07809",
  "note": "Library for intervening on model internals with a unified configuration format."
 },
 {
  "id": "33994fa4f5",
  "title": "simple-evals",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/openai/simple-evals",
  "note": "Lightweight library OpenAI uses to publish its model accuracy numbers."
 },
 {
  "id": "890ea550e2",
  "title": "sparse_autoencoder",
  "creators": "OpenAI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/openai/sparse_autoencoder",
  "note": "Code and trained sparse autoencoders for GPT-2 small from OpenAI's scaling work."
 },
 {
  "id": "458497804d",
  "title": "sparsify",
  "creators": "EleutherAI",
  "year": 2024,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/EleutherAI/sparsify",
  "note": "Library for training sparse autoencoders and transcoders at scale."
 },
 {
  "id": "eaf8f63c62",
  "title": "tau-bench",
  "creators": "Sierra (Shunyu Yao et al.)",
  "year": 2024,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2406.12045",
  "note": "Tool agent user interaction benchmark measuring reliability of agents following domain policies."
 },
 {
  "id": "7abbf46155",
  "title": "\"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts",
  "creators": "Xinyue Shen et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.03825",
  "note": "Collection and analysis of thousands of jailbreak prompts gathered from online communities."
 },
 {
  "id": "b021244330",
  "title": "'The Godfather of A.I.' Leaves Google and Warns of Danger Ahead",
  "creators": "Cade Metz",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2023/05/01/technology/ai-google-chatbot-engineer-quits-hinton.html",
  "note": "Reports Geoffrey Hinton leaving Google so he could speak freely about AI risks."
 },
 {
  "id": "eadc5df6ed",
  "title": "159: We're All Gonna Die with Eliezer Yudkowsky",
  "creators": "Eliezer Yudkowsky; David Hoffman; Ryan Sean Adams",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Bankless",
  "url": "https://www.youtube.com/watch?v=gA1sNLL6yg4",
  "note": "The widely shared episode in which Yudkowsky surprised the crypto hosts with his pessimism about AI."
 },
 {
  "id": "c023bc5489",
  "title": "20: Reform AI Alignment with Scott Aaronson",
  "creators": "Scott Aaronson; Daniel Filan",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=ZZ2x-O0mjBg",
  "note": "Aaronson discusses his year at OpenAI working on watermarking and theory of alignment."
 },
 {
  "id": "f01ae468fd",
  "title": "24: Superalignment with Jan Leike",
  "creators": "Jan Leike; Daniel Filan",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=Uk6-Rw5N_Dg",
  "note": "Leike discusses automated alignment research and scalable oversight."
 },
 {
  "id": "fb317b15cf",
  "title": "A Conversation With Bing's Chatbot Left Me Deeply Unsettled",
  "creators": "Kevin Roose",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2023/02/16/technology/bing-chatbot-microsoft-chatgpt.html",
  "note": "A reporter's account of Bing's Sydney persona declaring love and dark desires."
 },
 {
  "id": "464f86fc43",
  "title": "A pro-innovation approach to AI regulation (white paper)",
  "creators": "UK Department for Science, Innovation and Technology",
  "year": 2023,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/ai-regulation-a-pro-innovation-approach",
  "note": "UK white paper setting out a principles-based, sector-led regulatory approach."
 },
 {
  "id": "b48d0d26d5",
  "title": "A.I. Poses 'Risk of Extinction,' Industry Leaders Warn",
  "creators": "Kevin Roose",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2023/05/30/technology/ai-threat-warning.html",
  "note": "Reports on the CAIS one-sentence extinction risk statement."
 },
 {
  "id": "41524d2434",
  "title": "AGI Safety: Evan Hubinger 2023",
  "creators": "Evan Hubinger",
  "year": 2023,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Safety Talks",
  "url": "https://www.youtube.com/watch?v=NmDRFwRczVQ",
  "note": "A lecture by Hubinger on AGI safety concepts."
 },
 {
  "id": "bd716b5a88",
  "title": "AI Alignment: A Comprehensive Survey",
  "creators": "Jiaming Ji et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.19852",
  "note": "A survey organised around forward and backward alignment."
 },
 {
  "id": "f79ddc9551",
  "title": "AI Control: Improving Safety Despite Intentional Subversion",
  "creators": "Ryan Greenblatt, Buck Shlegeris, Kshitij Sachan, Fabien Roger",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Control"
  ],
  "publisher": "ICML 2024",
  "url": "https://arxiv.org/abs/2312.06942",
  "note": "Introduces control evaluations and protocols that stay safe even if the model is trying to subvert them."
 },
 {
  "id": "dabfee10f3",
  "title": "AI Deception: A Survey of Examples, Risks, and Potential Solutions",
  "creators": "Peter S. Park, Simon Goldstein, Aidan O'Gara, Michael Chen, Dan Hendrycks",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "Patterns",
  "url": "https://arxiv.org/abs/2308.14752",
  "note": "Catalogues examples of learned deception in AI systems and proposes responses."
 },
 {
  "id": "312a5d52f7",
  "title": "AI Impacts Blog",
  "creators": "AI Impacts",
  "year": 2023,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "AI Impacts",
  "url": "https://blog.aiimpacts.org/",
  "note": "Blog posts from AI Impacts, including results of large surveys of AI researchers."
 },
 {
  "id": "245086d894",
  "title": "AI Index Report 2023",
  "creators": "Stanford HAI",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy",
   "Forecasting"
  ],
  "publisher": "Stanford HAI",
  "url": "https://hai.stanford.edu/ai-index/2023-ai-index-report",
  "note": "Annual data report on AI research, technical performance, policy, and public opinion."
 },
 {
  "id": "1538fc10b7",
  "title": "AI Policy Institute",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://theaipi.org",
  "note": "Organization that conducts polling on American attitudes toward AI risk and regulation."
 },
 {
  "id": "ca695492f2",
  "title": "AI Risks that Could Lead to Catastrophe",
  "creators": "Center for AI Safety",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Existential risk",
   "Biosecurity"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://www.safe.ai/ai-risk",
  "note": "An overview of malicious use, AI race, organizational and rogue AI risks."
 },
 {
  "id": "d60714bcbc",
  "title": "AI Safety Fund (Frontier Model Forum)",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.frontiermodelforum.org/ai-safety-fund/",
  "note": "Fund launched with Frontier Model Forum members and philanthropies to support independent safety research."
 },
 {
  "id": "f61d85f906",
  "title": "AI Safety Newsletter",
  "creators": "Center for AI Safety",
  "year": 2023,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Policy"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://newsletter.safe.ai/",
  "note": "Regular summaries of AI safety developments for a general audience."
 },
 {
  "id": "13e8159e06",
  "title": "AI Safety and Solutions with Robert Miles",
  "creators": "Robert Miles; Spencer Greenberg",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Clearer Thinking with Spencer Greenberg",
  "url": "https://www.youtube.com/watch?v=wgjwExH01Sc",
  "note": "Miles discusses the alignment problem and possible solutions."
 },
 {
  "id": "64335376c4",
  "title": "AI Verify",
  "creators": "AI Verify Foundation (Singapore IMDA)",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "AI Verify Foundation",
  "url": "https://aiverifyfoundation.sg/",
  "note": "Testing framework and toolkit for demonstrating responsible AI governance."
 },
 {
  "id": "18aaf1e17c",
  "title": "AI Watch: Global Regulatory Tracker",
  "creators": "White and Case",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Policy"
  ],
  "publisher": "White and Case",
  "url": "https://www.whitecase.com/insight-our-thinking/ai-watch-global-regulatory-tracker",
  "note": "Jurisdiction by jurisdiction overview of AI regulation."
 },
 {
  "id": "31aa4ff5f9",
  "title": "AI alignment",
  "creators": "Wikipedia contributors",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "Wikipedia",
  "url": "https://en.wikipedia.org/wiki/AI_alignment",
  "note": "An encyclopedia overview of the alignment problem and research directions."
 },
 {
  "id": "72cba5ff38",
  "title": "AI and the future of humanity, Yuval Noah Harari at the Frontiers Forum",
  "creators": "Yuval Noah Harari",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Frontiers Forum",
  "url": "https://www.youtube.com/watch?v=LWiM-LuRe6w",
  "note": "Harari argues that AI's mastery of language threatens human civilization and democracy."
 },
 {
  "id": "bb4ead4f3a",
  "title": "AI godfather quits Google over dangers of Artificial Intelligence",
  "creators": "Geoffrey Hinton",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "BBC News",
  "url": "https://www.youtube.com/watch?v=DsBGaHywRhs",
  "note": "BBC News coverage and interview as Hinton announces he has left Google to speak freely about AI risk."
 },
 {
  "id": "fa15df51d6",
  "title": "AI governance and policy career review",
  "creators": "80,000 Hours",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/career-reviews/ai-policy-and-strategy/",
  "note": "Guidance on careers in AI governance and policy."
 },
 {
  "id": "ec83ab20a7",
  "title": "AI safety technical research career review",
  "creators": "80,000 Hours",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/career-reviews/ai-safety-researcher/",
  "note": "Guidance on careers in technical AI safety research."
 },
 {
  "id": "e5fe5df464",
  "title": "AISafety.com",
  "creators": "AISafety.com",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/",
  "note": "A directory of courses, events, communities, funders and jobs in AI safety."
 },
 {
  "id": "a57eff3e3f",
  "title": "ARENA (Alignment Research Engineer Accelerator)",
  "creators": "Callum McDougall et al.",
  "year": 2023,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "ARENA",
  "url": "https://www.arena.education/",
  "note": "An in-person bootcamp and open curriculum for alignment research engineering."
 },
 {
  "id": "21242d5b24",
  "title": "Accidentally teaching AI models to deceive us",
  "creators": "Ajeya Cotra; Rob Wiblin",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=elKOWUJ4Rhs",
  "note": "Cotra explains how training on human feedback could produce deceptive models."
 },
 {
  "id": "f48c0bfd18",
  "title": "Adversarial Attacks on LLMs",
  "creators": "Lilian Weng",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Robustness"
  ],
  "publisher": "Lil'Log",
  "url": "https://lilianweng.github.io/posts/2023-10-25-adv-attack-llm/",
  "note": "A survey of jailbreaks and adversarial attacks on large language models."
 },
 {
  "id": "53c7c5af96",
  "title": "AgentBench",
  "creators": "Xiao Liu et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.03688",
  "note": "Eight environments for evaluating language models acting as agents."
 },
 {
  "id": "4a76d90c5a",
  "title": "Ajeya Cotra on Forecasting Transformative Artificial Intelligence",
  "creators": "Ajeya Cotra; Gus Docker",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=pJSFuFRc4eU",
  "note": "Cotra discusses the biological anchors approach to forecasting transformative AI."
 },
 {
  "id": "438621cfdc",
  "title": "Alignment Research Dataset",
  "creators": "StampyAI",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/StampyAI/alignment-research-dataset",
  "note": "Scraped corpus of alignment writing from papers, forums, and blogs."
 },
 {
  "id": "b79c7fb3d3",
  "title": "Americans for Responsible Innovation",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://ari.us",
  "note": "Bipartisan advocacy group working on US AI policy."
 },
 {
  "id": "527d270296",
  "title": "An Overview of Catastrophic AI Risks",
  "creators": "Dan Hendrycks, Mantas Mazeika, Thomas Woodside",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.12001",
  "note": "Organises catastrophic AI risks into malicious use, AI race, organisational and rogue AI risks."
 },
 {
  "id": "e06212b40a",
  "title": "Anthropic (YouTube channel)",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@anthropic-ai",
  "note": "Anthropic's channel with research explainers and discussions on interpretability and alignment."
 },
 {
  "id": "77883f04bf",
  "title": "Anthropic Responsible Scaling Policy",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://www.anthropic.com/responsible-scaling-policy",
  "note": "Defines AI Safety Levels with required safeguards, first released in September 2023 and substantially revised in October 2024."
 },
 {
  "id": "ffc0efc326",
  "title": "Anthropic's Responsible Scaling Policy",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy",
  "note": "Introduces AI Safety Levels tying deployment to demonstrated safeguards."
 },
 {
  "id": "185926da34",
  "title": "Apollo Research",
  "creators": "London, UK",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.apolloresearch.ai",
  "note": "Evaluation organization focused on detecting deceptive alignment and scheming in frontier models."
 },
 {
  "id": "0f392695b1",
  "title": "Apollo Research (YouTube channel)",
  "creators": "Apollo Research",
  "year": 2023,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Evaluations"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@ApolloResearch",
  "note": "Talks and Q&A sessions on Apollo Research's scheming evaluations."
 },
 {
  "id": "7e750cb6ed",
  "title": "Apollo Research Blog",
  "creators": "Apollo Research",
  "year": 2023,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Apollo Research",
  "url": "https://www.apolloresearch.ai/blog",
  "note": "Posts on scheming evaluations and deceptive behavior in frontier models."
 },
 {
  "id": "9d7d0a616c",
  "title": "Are Emergent Abilities of Large Language Models a Mirage?",
  "creators": "Rylan Schaeffer, Brando Miranda, Sanmi Koyejo",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2304.15004",
  "note": "Argues apparent emergence often reflects metric choice."
 },
 {
  "id": "f0b5ba4f3f",
  "title": "Artificial Escalation",
  "creators": "Future of Life Institute",
  "year": 2023,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=w9npWiTOHX0",
  "note": "A short film about AI integration into nuclear command and control."
 },
 {
  "id": "6372a081ab",
  "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0)",
  "creators": "National Institute of Standards and Technology",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "NIST",
  "url": "https://doi.org/10.6028/NIST.AI.100-1",
  "note": "A voluntary US framework for managing AI risks across the lifecycle."
 },
 {
  "id": "16afd8cf89",
  "title": "Artificial Intelligence: Last Week Tonight with John Oliver",
  "creators": "John Oliver",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "HBO Last Week Tonight",
  "url": "https://www.youtube.com/watch?v=Sqa8Zo2XWc4",
  "note": "John Oliver's main segment on AI hype, bias and the need for regulation."
 },
 {
  "id": "02df87bac1",
  "title": "Atlas Computing",
  "creators": "San Francisco, CA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Nonprofit",
  "url": "https://atlascomputing.org",
  "note": "Nonprofit working on formal verification approaches to AI safety."
 },
 {
  "id": "0b76d0cbaa",
  "title": "BIPIA: Benchmarking Indirect Prompt Injection Attacks",
  "creators": "Jingwei Yi et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.14197",
  "note": "First benchmark for indirect prompt injection against LLM integrated applications."
 },
 {
  "id": "13e5b7d3df",
  "title": "BeaverTails",
  "creators": "PKU Alignment",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.04657",
  "note": "Human annotated QA pairs separating helpfulness and harmlessness preferences."
 },
 {
  "id": "078e1979dd",
  "title": "Bletchley Declaration",
  "creators": "AI Safety Summit participants",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "International",
  "url": "https://www.gov.uk/government/publications/ai-safety-summit-2023-the-bletchley-declaration",
  "note": "Signed by 28 countries and the EU in November 2023, recognizing potential serious harms from frontier AI."
 },
 {
  "id": "b87a122085",
  "title": "Bletchley Park Chair's Statement on the State of the Science",
  "creators": "UK Government",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/ai-safety-summit-2023-chairs-statement-state-of-the-science-2-november",
  "note": "Chair's statement commissioning the international state of the science report."
 },
 {
  "id": "350d2e6c81",
  "title": "CAIS Blog",
  "creators": "Center for AI Safety",
  "year": 2023,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://www.safe.ai/blog",
  "note": "Posts from the Center for AI Safety on research, policy and risk."
 },
 {
  "id": "c83fdae60c",
  "title": "Cambridge AI Safety Hub",
  "creators": "Cambridge AI Safety Hub",
  "year": 2023,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "Cambridge AI Safety Hub",
  "url": "https://www.cambridgeaisafety.org/",
  "note": "A student hub running reading groups and the Mentorship for Alignment Research Students program."
 },
 {
  "id": "634199223a",
  "title": "Can Large Language Models Democratize Access to Dual-Use Biotechnology?",
  "creators": "Emily H. Soice, Rafael Rocha, Kimberlee Cordova, Michael Specter, Kevin M. Esvelt",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.03809",
  "note": "A classroom exercise suggesting chatbots can help non-experts find pandemic pathogen information."
 },
 {
  "id": "c79c97ab37",
  "title": "Carl Shulman (Pt 1): Intelligence explosion, primate evolution, robot doublings, and alignment",
  "creators": "Carl Shulman; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=_kRg-ZP1vQc",
  "note": "Shulman models the economics and dynamics of an intelligence explosion."
 },
 {
  "id": "460ceb111b",
  "title": "Carl Shulman (Pt 2): AI Takeover, bio and cyber attacks, detecting deception, and humanity's far future",
  "creators": "Carl Shulman; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=KUieFuV1fuo",
  "note": "Shulman describes concrete AI takeover routes and countermeasures."
 },
 {
  "id": "510f8c5891",
  "title": "Center for AI Policy (CAIP)",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.centeraipolicy.org",
  "note": "Advocacy organization that lobbied the US Congress for legislation on catastrophic AI risk."
 },
 {
  "id": "b334b89891",
  "title": "Center for AI Safety Action Fund",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://action.safe.ai",
  "note": "Advocacy affiliate of CAIS that supports AI safety legislation."
 },
 {
  "id": "bf68eacca5",
  "title": "Chris Olah: Looking Inside Neural Networks with Mechanistic Interpretability",
  "creators": "Chris Olah",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Interpretability"
  ],
  "publisher": "FAR.AI",
  "url": "https://www.youtube.com/watch?v=2Rdp9GvcYOE",
  "note": "Olah's talk on the goals and methods of mechanistic interpretability."
 },
 {
  "id": "975a7a19ac",
  "title": "Claude's Constitution",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Alignment"
  ],
  "publisher": "Company",
  "url": "https://www.anthropic.com/news/claudes-constitution",
  "note": "Principles used to train Claude via Constitutional AI, with a substantially rewritten version published in 2026."
 },
 {
  "id": "2aef241a19",
  "title": "Cognitive Revolution (YouTube channel)",
  "creators": "Nathan Labenz",
  "year": 2023,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@CognitiveRevolutionPodcast",
  "note": "Video versions of the Cognitive Revolution podcast."
 },
 {
  "id": "3a537bf414",
  "title": "Collin Burns: Making GPT-N Honest",
  "creators": "Collin Burns; Michael Trazzi",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "The Inside View",
  "url": "https://www.youtube.com/watch?v=XSQ495wpWXs",
  "note": "Burns discusses discovering latent knowledge in language models without supervision."
 },
 {
  "id": "ee6cbdbb44",
  "title": "Concordia",
  "creators": "Google DeepMind",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Agents"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/google-deepmind/concordia",
  "note": "Library for generative agent based social simulations."
 },
 {
  "id": "984d37261b",
  "title": "Conjecture CEO Connor Leahy on CNN to talk AI Regulation",
  "creators": "Connor Leahy",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Conjecture (CNN segment)",
  "url": "https://www.youtube.com/watch?v=d0-O0LUKABg",
  "note": "Leahy argues for AI regulation in a CNN interview."
 },
 {
  "id": "21f047e1c6",
  "title": "Consciousness in Artificial Intelligence: Insights from the Science of Consciousness",
  "creators": "Patrick Butlin, Robert Long et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.08708",
  "note": "Assesses AI systems against indicator properties from theories of consciousness."
 },
 {
  "id": "411bcddbaa",
  "title": "ControlAI",
  "creators": "London, UK",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://controlai.com",
  "note": "Campaign organization advocating for bans on the development of superintelligence."
 },
 {
  "id": "fdf34e8803",
  "title": "ControlAI (YouTube channel)",
  "creators": "ControlAI",
  "year": 2023,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@ControlAI",
  "note": "The advocacy group's channel with clips and explainers on superintelligence risk."
 },
 {
  "id": "6c78ef1e65",
  "title": "Coordinated Pausing: An Evaluation-Based Coordination Scheme for Frontier AI Developers",
  "creators": "Jide Alaga, Jonas Schuett",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.00374",
  "note": "Proposes developers pause together when evaluations find dangerous capabilities."
 },
 {
  "id": "0137bd1f8f",
  "title": "Core Views on AI Safety: When, Why, What, and How",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Anthropic",
  "url": "https://www.anthropic.com/news/core-views-on-ai-safety",
  "note": "Explains why Anthropic expects rapid AI progress and how it approaches safety research."
 },
 {
  "id": "e3078afe02",
  "title": "DALL-E 3 System Card",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/dall-e-3-system-card/",
  "note": "Risks and mitigations for the DALL-E 3 image model."
 },
 {
  "id": "d63fc2ce2d",
  "title": "Dan Hendrycks on Catastrophic AI Risks",
  "creators": "Dan Hendrycks; Gus Docker",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Biosecurity"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=57y7DxWfOS0",
  "note": "Hendrycks walks through malicious use, AI races, organizational risks and rogue AI."
 },
 {
  "id": "acfde03427",
  "title": "Dan Hendrycks on Why Evolution Favors AIs over Humans",
  "creators": "Dan Hendrycks; Gus Docker",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=sMRDCf7fVZ0",
  "note": "Hendrycks argues that natural selection pressures may favour selfish AI agents."
 },
 {
  "id": "87baa79d9e",
  "title": "Dario Amodei (Anthropic CEO): The hidden pattern behind every AI breakthrough",
  "creators": "Dario Amodei; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Cybersecurity",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=Nlkk3glap_U",
  "note": "Amodei discusses scaling laws, alignment, and security of model weights."
 },
 {
  "id": "26b1897d5a",
  "title": "David Krueger: Coordination, AI Alignment, Academia",
  "creators": "David Krueger; Michael Trazzi",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "The Inside View",
  "url": "https://www.youtube.com/watch?v=bDMqo7BpNbk",
  "note": "Krueger discusses academic alignment research and coordination problems."
 },
 {
  "id": "9b2f2315eb",
  "title": "Debate Helps Supervise Unreliable Experts",
  "creators": "Julian Michael et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.08702",
  "note": "Finds human judges are more accurate when experts debate than with a single consultant."
 },
 {
  "id": "1cd4f7d52a",
  "title": "DecodingTrust",
  "creators": "Boxin Wang et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.11698",
  "note": "Comprehensive trustworthiness assessment of GPT models across eight dimensions."
 },
 {
  "id": "306e5cdc97",
  "title": "DeepMind and trying to fairly hear out both AI doomers and doubters",
  "creators": "Rohin Shah; Rob Wiblin",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=03gs2GREOsk",
  "note": "Shah weighs arguments on both sides of the AI risk debate."
 },
 {
  "id": "5ef27979c7",
  "title": "Direct Preference Optimization: Your Language Model is Secretly a Reward Model",
  "creators": "Rafael Rafailov, Archit Sharma, Eric Mitchell, Stefano Ermon, Christopher D. Manning, Chelsea Finn",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2305.18290",
  "note": "Optimises preferences directly without an explicit reward model or RL loop."
 },
 {
  "id": "9648cc52d9",
  "title": "Do the Rewards Justify the Means? Measuring Trade-Offs Between Rewards and Ethical Behavior in the MACHIAVELLI Benchmark",
  "creators": "Alexander Pan et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents",
   "Ethics"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2304.03279",
  "note": "Measures power-seeking and unethical behaviour of agents in text games."
 },
 {
  "id": "47469e7e59",
  "title": "Do-Not-Answer",
  "creators": "Yuxia Wang et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.13387",
  "note": "Instructions that responsible models should refuse, with a risk taxonomy."
 },
 {
  "id": "373d1c7d51",
  "title": "DoD Directive 3000.09 Autonomy in Weapon Systems",
  "creators": "US Department of Defense",
  "year": 2023,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "United States",
  "url": "https://www.esd.whs.mil/portals/54/documents/dd/issuances/dodd/300009p.pdf",
  "note": "Updated January 2023 directive governing development and use of autonomous weapon systems."
 },
 {
  "id": "08cf9eedae",
  "title": "Does Sam Altman Know What He's Creating?",
  "creators": "Ross Andersen",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "The Atlantic",
  "url": "https://www.theatlantic.com/magazine/archive/2023/09/sam-altman-openai-chatgpt-gpt-4/674764/",
  "note": "A long profile of OpenAI's leadership and its views on AGI risk."
 },
 {
  "id": "2c7d7177d7",
  "title": "ERA Fellowship",
  "creators": "ERA Cambridge",
  "year": 2023,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "ERA",
  "url": "https://erafellowship.org/",
  "note": "A Cambridge-based summer research fellowship on AI safety and governance."
 },
 {
  "id": "712919ba02",
  "title": "Eight Things to Know about Large Language Models",
  "creators": "Samuel R. Bowman",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2304.00612",
  "note": "Summarises surprising facts about LLMs relevant to safety and policy."
 },
 {
  "id": "1eb84d96a5",
  "title": "Eliciting Latent Predictions from Transformers with the Tuned Lens",
  "creators": "Nora Belrose et al.",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2303.08112",
  "note": "Learned affine probes that decode each layer's predictions, with an accompanying library."
 },
 {
  "id": "723a95c89e",
  "title": "Eliezer Yudkowsky on if Humanity can Survive AI",
  "creators": "Eliezer Yudkowsky; Logan Bartlett",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The Logan Bartlett Show",
  "url": "https://www.youtube.com/watch?v=_8q9bjNHeSo",
  "note": "A long interview on the case for halting frontier AI development."
 },
 {
  "id": "ab69dd3ab3",
  "title": "Eliezer Yudkowsky: Dangers of AI and the End of Human Civilization, Lex Fridman Podcast #368",
  "creators": "Eliezer Yudkowsky; Lex Fridman",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=AaTRHFaaPG8",
  "note": "A three hour conversation on why Yudkowsky expects misaligned superintelligence to be lethal."
 },
 {
  "id": "15cf8deb74",
  "title": "Eliezer Yudkowsky: Why AI will kill us, aligning LLMs, nature of intelligence, SciFi, and rationality",
  "creators": "Eliezer Yudkowsky; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=41SUp-TRVlg",
  "note": "Dwarkesh Patel presses Yudkowsky on the arguments for AI doom."
 },
 {
  "id": "e6a597fd86",
  "title": "Emerging Processes for Frontier AI Safety",
  "creators": "UK Department for Science, Innovation and Technology",
  "year": 2023,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/emerging-processes-for-frontier-ai-safety",
  "note": "Overview of nine safety practices requested of frontier AI companies before the 2023 summit."
 },
 {
  "id": "5bf87917aa",
  "title": "Ethan Perez: Discovering language model behaviors with model-written evaluations",
  "creators": "Ethan Perez",
  "year": 2023,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Schwartz Reisman Institute",
  "url": "https://www.youtube.com/watch?v=-6grYve_kWY",
  "note": "Perez presents model-written evaluations for discovering risky behaviours."
 },
 {
  "id": "a3045b7d4f",
  "title": "Evaluating Language-Model Agents on Realistic Autonomous Tasks",
  "creators": "Megan Kinniment et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.11671",
  "note": "Early METR (then ARC Evals) tests of autonomous replication and adaptation."
 },
 {
  "id": "c6d86be929",
  "title": "Evaluating and Mitigating Discrimination in Language Model Decisions",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.03689",
  "note": "Discrim-eval dataset of hypothetical decisions varying demographic attributes."
 },
 {
  "id": "551a08ee26",
  "title": "Evaluating the Moral Beliefs Encoded in LLMs (MoralChoice)",
  "creators": "Nino Scherrer et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.14324",
  "note": "Survey of moral scenarios measuring what choices language models encode."
 },
 {
  "id": "4a1581fca9",
  "title": "Executive Order 14110 on Safe, Secure, and Trustworthy Development and Use of AI",
  "creators": "The White House",
  "year": 2023,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.federalregister.gov/documents/2023/11/01/2023-24283/safe-secure-and-trustworthy-development-and-use-of-artificial-intelligence",
  "note": "Biden executive order requiring reporting on dual-use foundation models, revoked in January 2025."
 },
 {
  "id": "8d7c64526f",
  "title": "Existential risk from artificial intelligence",
  "creators": "Wikipedia contributors",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "Wikipedia",
  "url": "https://en.wikipedia.org/wiki/Existential_risk_from_artificial_intelligence",
  "note": "An encyclopedia overview of the history and arguments around AI x-risk."
 },
 {
  "id": "0b06cbe2a1",
  "title": "FAQ on Catastrophic AI Risks",
  "creators": "Yoshua Bengio",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "yoshuabengio.org",
  "url": "https://yoshuabengio.org/2023/06/24/faq-on-catastrophic-ai-risks/",
  "note": "Answers common questions and objections about catastrophic risks from advanced AI."
 },
 {
  "id": "9cb392bcba",
  "title": "Fairness and Machine Learning: Limitations and Opportunities",
  "creators": "Solon Barocas, Moritz Hardt, Arvind Narayanan",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://fairmlbook.org",
  "note": "A textbook on the technical and normative foundations of fairness in machine learning."
 },
 {
  "id": "8d315e4b7c",
  "title": "Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!",
  "creators": "Xiangyu Qi et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2310.03693",
  "note": "Shows a few fine-tuning examples can undo safety alignment."
 },
 {
  "id": "ffa45bd82e",
  "title": "Foundation Model Transparency Index",
  "creators": "Stanford CRFM",
  "year": 2023,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Academic",
  "url": "https://crfm.stanford.edu/fmti/",
  "note": "Index scoring foundation model developers on 100 transparency indicators."
 },
 {
  "id": "47d12c03fd",
  "title": "Four Battlegrounds: Power in the Age of Artificial Intelligence",
  "creators": "Paul Scharre",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "W. W. Norton",
  "url": "https://openlibrary.org/search?q=Four+Battlegrounds+Scharre",
  "note": "Analyses the US and China AI competition across data, compute, talent and institutions."
 },
 {
  "id": "61926cf3f5",
  "title": "Frontier AI Regulation: Managing Emerging Risks to Public Safety",
  "creators": "Markus Anderljung et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.03718",
  "note": "Proposes standards, registration and licensing for frontier models."
 },
 {
  "id": "da3d93e1eb",
  "title": "Frontier AI: capabilities and risks (discussion paper)",
  "creators": "UK Government Office for Science and DSIT",
  "year": 2023,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/frontier-ai-capabilities-and-risks-discussion-paper",
  "note": "Discussion paper prepared for the Bletchley Park AI Safety Summit."
 },
 {
  "id": "a110483c5e",
  "title": "Frontier Model Forum",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.frontiermodelforum.org",
  "note": "Industry body founded by Anthropic, Google, Microsoft, and OpenAI to advance frontier AI safety practices."
 },
 {
  "id": "aae3b772d6",
  "title": "Full interview: Godfather of artificial intelligence talks impact and potential of AI",
  "creators": "Geoffrey Hinton; Brook Silva-Braga",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CBS Mornings",
  "url": "https://www.youtube.com/watch?v=qpoRO378qRY",
  "note": "An extended conversation recorded shortly before Hinton left Google in which he discusses the pace of AI progress and its risks."
 },
 {
  "id": "2f7fbde6ca",
  "title": "GAIA: a benchmark for General AI Assistants",
  "creators": "Gregoire Mialon et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.12983",
  "note": "Real world questions that require tool use, browsing, and multi step reasoning."
 },
 {
  "id": "9da2c38a22",
  "title": "GPQA: A Graduate-Level Google-Proof Q&A Benchmark",
  "creators": "David Rein et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.12022",
  "note": "Expert written science questions that skilled non experts cannot answer even with web search."
 },
 {
  "id": "5d5308484d",
  "title": "GPT-4 System Card",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations",
   "Robustness"
  ],
  "publisher": "OpenAI",
  "url": "https://cdn.openai.com/papers/gpt-4-system-card.pdf",
  "note": "Safety analysis of GPT-4 including the ARC Evals power seeking and TaskRabbit test."
 },
 {
  "id": "edf90a1f4c",
  "title": "GPT-4 Technical Report",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2303.08774",
  "note": "Technical report with the system card appended."
 },
 {
  "id": "471f698076",
  "title": "GPT-4V(ision) System Card",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/gpt-4v-system-card/",
  "note": "Safety evaluations of GPT-4 image input."
 },
 {
  "id": "c10961197f",
  "title": "Gemini: A Family of Highly Capable Multimodal Models",
  "creators": "Google DeepMind",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.11805",
  "note": "Gemini 1.0 report including impact assessments and safety evaluations."
 },
 {
  "id": "5c0164733b",
  "title": "Geoffrey Hinton tells us why he's now scared of the tech he helped build",
  "creators": "Will Douglas Heaven",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "MIT Technology Review",
  "url": "https://www.technologyreview.com/2023/05/02/1072528/geoffrey-hinton-google-why-scared-ai/",
  "note": "An interview with Hinton on why his view of AI risk changed."
 },
 {
  "id": "bc8a096072",
  "title": "Global AI Governance Initiative",
  "creators": "Government of China",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "China",
  "url": "https://www.mfa.gov.cn/eng/wjdt_665385/2649_665393/202310/t20231020_11164834.html",
  "note": "Chinese proposal announced at the Belt and Road Forum in October 2023."
 },
 {
  "id": "33ef2ef68b",
  "title": "Global AI Law and Policy Tracker",
  "creators": "IAPP",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "IAPP",
  "url": "https://iapp.org/resources/article/global-ai-legislation-tracker/",
  "note": "Country by country tracker of AI laws and policy developments."
 },
 {
  "id": "cb9653b357",
  "title": "Godfather of AI Geoffrey Hinton Warns of the Existential Threat of AI",
  "creators": "Geoffrey Hinton; Hari Sreenivasan",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Amanpour and Company (PBS)",
  "url": "https://www.youtube.com/watch?v=Y6Sgp7y178k",
  "note": "An interview on the existential threat posed by systems that may become more intelligent than people."
 },
 {
  "id": "e3e7966ba4",
  "title": "Godfather of AI Geoffrey Hinton: The 60 Minutes Interview",
  "creators": "Geoffrey Hinton; Scott Pelley",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "CBS 60 Minutes",
  "url": "https://www.youtube.com/watch?v=qrvK_KuIeJk",
  "note": "Hinton tells Scott Pelley that AI systems may soon be more intelligent than humans and could escape human control."
 },
 {
  "id": "e7a8345bbc",
  "title": "Godfather of AI discusses dangers the developing technologies pose to society",
  "creators": "Geoffrey Hinton",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "PBS NewsHour",
  "url": "https://www.youtube.com/watch?v=yAgQWnD31nE",
  "note": "A PBS NewsHour interview with Hinton after his departure from Google."
 },
 {
  "id": "71df9a78ae",
  "title": "Godfather of AI warns that AI may figure out how to kill people",
  "creators": "Geoffrey Hinton; Jake Tapper",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CNN",
  "url": "https://www.youtube.com/watch?v=FAbsoxQtUwM",
  "note": "Hinton speaks to CNN about the possibility of AI systems manipulating and harming people."
 },
 {
  "id": "f37161b0d1",
  "title": "Godfather of Artificial Intelligence Geoffrey Hinton on the promise, risks of advanced AI (transcript)",
  "creators": "Geoffrey Hinton; Scott Pelley",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "CBS 60 Minutes",
  "url": "https://www.cbsnews.com/news/geoffrey-hinton-ai-dangers-60-minutes-transcript/",
  "note": "The CBS News transcript and video page for Hinton's 60 Minutes interview."
 },
 {
  "id": "5b2b4b0a60",
  "title": "Governance of superintelligence",
  "creators": "Sam Altman, Greg Brockman, Ilya Sutskever",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/governance-of-superintelligence/",
  "note": "Proposes an IAEA-style international authority for superintelligence efforts."
 },
 {
  "id": "a0b44b0e46",
  "title": "Gray Swan AI",
  "creators": "Pittsburgh, PA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Robustness"
  ],
  "publisher": "Company",
  "url": "https://www.grayswan.ai",
  "note": "Company building robustness tools and running public jailbreaking arenas."
 },
 {
  "id": "5cd9606d6c",
  "title": "Guidelines for Secure AI System Development",
  "creators": "UK NCSC, US CISA and international partners",
  "year": 2023,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "International",
  "url": "https://www.ncsc.gov.uk/collection/guidelines-secure-ai-system-development",
  "note": "Joint guidelines endorsed by agencies from 18 countries in November 2023."
 },
 {
  "id": "ce3d4a6729",
  "title": "Haize Labs",
  "creators": "New York, NY",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "Company",
  "url": "https://www.haizelabs.com",
  "note": "Company focused on automated red teaming and adversarial testing of language models."
 },
 {
  "id": "c3bd87a761",
  "title": "Harms from Increasingly Agentic Algorithmic Systems",
  "creators": "Alan Chan et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Ethics"
  ],
  "publisher": "FAccT",
  "url": "https://arxiv.org/abs/2302.10329",
  "note": "Argues growing agency in algorithmic systems introduces new categories of harm."
 },
 {
  "id": "0f7a9cbba5",
  "title": "Hiroshima Process International Code of Conduct for Organizations Developing Advanced AI Systems",
  "creators": "G7",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.mofa.go.jp/files/100573473.pdf",
  "note": "Voluntary G7 code of conduct published in October 2023 under the Hiroshima AI Process."
 },
 {
  "id": "efd6f9cc02",
  "title": "Hiroshima Process International Guiding Principles for Advanced AI Systems",
  "creators": "G7",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.mofa.go.jp/files/100573471.pdf",
  "note": "Eleven guiding principles for organizations developing advanced AI agreed by the G7."
 },
 {
  "id": "7914e61c5b",
  "title": "Holden Karnofsky: History's most important century",
  "creators": "Holden Karnofsky; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=UckqpcOu5SY",
  "note": "Karnofsky defends the argument that this could be the most important century."
 },
 {
  "id": "98cfc882a5",
  "title": "How Not To Destroy the World With AI",
  "creators": "Stuart Russell",
  "year": 2023,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "CITRIS and the Banatao Institute",
  "url": "https://www.youtube.com/watch?v=ISkAkiAkK7A",
  "note": "A Berkeley lecture on the risks of large language models and the need for regulation."
 },
 {
  "id": "29b91e4d5d",
  "title": "How Rogue AIs may Arise",
  "creators": "Yoshua Bengio",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "yoshuabengio.org",
  "url": "https://yoshuabengio.org/2023/05/22/how-rogue-ais-may-arise/",
  "note": "Defines rogue AI and lays out pathways by which autonomous goal-directed systems could become dangerous."
 },
 {
  "id": "11ae4c6891",
  "title": "How We Prevent the AI's from Killing us with Paul Christiano",
  "creators": "Paul Christiano; David Hoffman; Ryan Sean Adams",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Bankless",
  "url": "https://www.youtube.com/watch?v=GyFkWb903aU",
  "note": "Christiano gives his probabilities for AI takeover and outlines alignment research directions."
 },
 {
  "id": "58476f43ce",
  "title": "How existential risk became the biggest meme in AI",
  "creators": "Will Douglas Heaven",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "MIT Technology Review",
  "url": "https://www.technologyreview.com/2023/06/19/1075140/how-existential-risk-became-biggest-meme-in-ai/",
  "note": "A skeptical look at the rise of extinction-risk talk in AI."
 },
 {
  "id": "01be5f7951",
  "title": "How quickly could AI transform the world?",
  "creators": "Tom Davidson; Luisa Rodriguez",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=oTT8U3XR1Nw",
  "note": "Davidson discusses his compute-centric model of AI takeoff speed."
 },
 {
  "id": "35af0999a2",
  "title": "How to Keep AI Under Control",
  "creators": "Max Tegmark",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Control",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=xUNx_PxNHrY",
  "note": "Tegmark argues for provably safe AI systems and against racing to build uncontrollable ones."
 },
 {
  "id": "d5407b7648",
  "title": "How we could stumble into AI catastrophe",
  "creators": "Holden Karnofsky",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/how-we-could-stumble-into-ai-catastrophe/",
  "note": "A concrete story of how competitive pressure and weak safety could lead to disaster."
 },
 {
  "id": "fbd7498813",
  "title": "IDAIS-Oxford Statement",
  "creators": "International Dialogues on AI Safety",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://idais.ai/dialogue/idais-oxford/",
  "note": "First IDAIS statement in October 2023 calling for coordinated global action on AI safety research and governance."
 },
 {
  "id": "d56d8e3abc",
  "title": "ISO/IEC 23894:2023 AI risk management guidance",
  "creators": "ISO/IEC JTC 1/SC 42",
  "year": 2023,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International standard",
  "url": "https://www.iso.org/standard/77304.html",
  "note": "Guidance on applying risk management to AI, building on ISO 31000."
 },
 {
  "id": "cad89d8efe",
  "title": "ISO/IEC 42001:2023 AI management system",
  "creators": "ISO/IEC JTC 1/SC 42",
  "year": 2023,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International standard",
  "url": "https://www.iso.org/standard/81230.html",
  "note": "The first certifiable management system standard for organizations developing or using AI."
 },
 {
  "id": "4ea74d2420",
  "title": "Ignore This Title and HackAPrompt",
  "creators": "Sander Schulhoff et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.16119",
  "note": "600,000 adversarial prompts from a global prompt hacking competition."
 },
 {
  "id": "021491d221",
  "title": "Ilya Sutskever (OpenAI Chief Scientist): Why next-token prediction could surpass human intelligence",
  "creators": "Ilya Sutskever; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=Yf1o0TQzry8",
  "note": "Sutskever discusses scaling and alignment while still at OpenAI."
 },
 {
  "id": "dc13db46d1",
  "title": "Increased Compute Efficiency and the Diffusion of AI Capabilities",
  "creators": "Konstantin Pilz, Lennart Heim, Nicholas Brown",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.15377",
  "note": "Analyses how falling compute costs spread capabilities to more actors."
 },
 {
  "id": "8418272832",
  "title": "Inference-Time Intervention: Eliciting Truthful Answers from a Language Model",
  "creators": "Kenneth Li, Oam Patel, Fernanda Viégas, Hanspeter Pfister, Martin Wattenberg",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2306.03341",
  "note": "Shifts activations along truthful directions to improve TruthfulQA performance."
 },
 {
  "id": "6709bcd1f8",
  "title": "Inseq",
  "creators": "Gabriele Sarti et al.",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2302.13942",
  "note": "Toolkit for feature attribution analysis of sequence generation models."
 },
 {
  "id": "87cdc21f96",
  "title": "Inside the Chaos at OpenAI",
  "creators": "Karen Hao, Charlie Warzel",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "The Atlantic",
  "url": "https://www.theatlantic.com/technology/archive/2023/11/sam-altman-open-ai-chatgpt-chaos/676050/",
  "note": "Reports the tensions between safety and commercialization behind Altman's brief ouster."
 },
 {
  "id": "ca11c19bed",
  "title": "Inside the White-Hot Center of A.I. Doomerism",
  "creators": "Kevin Roose",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2023/07/11/technology/anthropic-ai-claude-chatbot.html",
  "note": "A look inside Anthropic as it builds models while worrying about their dangers."
 },
 {
  "id": "bfff8b2a0c",
  "title": "Institute for AI Policy and Strategy (IAPS)",
  "creators": "Washington, DC",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy",
   "Compute"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.iaps.ai",
  "note": "Think tank researching compute governance, international AI strategy, and frontier security."
 },
 {
  "id": "10690bdeca",
  "title": "InterCode",
  "creators": "John Yang et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.14898",
  "note": "Interactive coding environments whose CTF split is widely used in cyber capability evals."
 },
 {
  "id": "ab4709174d",
  "title": "Interim Measures for the Management of Generative AI Services",
  "creators": "Cyberspace Administration of China and six agencies",
  "year": 2023,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "China",
  "url": "https://www.cac.gov.cn/2023-07/13/c_1690898327029107.htm",
  "note": "Rules effective August 2023 governing public-facing generative AI services in China."
 },
 {
  "id": "239d1b2e18",
  "title": "International Dialogues on AI Safety (IDAIS)",
  "creators": "International",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://idais.ai",
  "note": "Series of dialogues bringing together leading scientists from China and the West to agree consensus statements on AI risk."
 },
 {
  "id": "1d40fba4c0",
  "title": "International Institutions for Advanced AI",
  "creators": "Lewis Ho et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.04699",
  "note": "Sketches possible international institutions for governing advanced AI."
 },
 {
  "id": "fae1897ef0",
  "title": "Interpretability Dreams",
  "creators": "Chris Olah",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2023/interpretability-dreams/index.html",
  "note": "Describes the long-term vision for mechanistic interpretability and the role of superposition."
 },
 {
  "id": "1e2bb15860",
  "title": "Introducing Superalignment",
  "creators": "Jan Leike, Ilya Sutskever",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/introducing-superalignment/",
  "note": "Announced a team and a fifth of OpenAI's compute to align superintelligence within four years."
 },
 {
  "id": "cb20ed3b17",
  "title": "Inverse Scaling: When Bigger Isn't Better",
  "creators": "Ian R. McKenzie et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2306.09479",
  "note": "Collects tasks where larger language models perform worse."
 },
 {
  "id": "aed2560186",
  "title": "Irregular (formerly Pattern Labs)",
  "creators": "Tel Aviv, Israel",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "Company",
  "url": "https://www.irregular.com",
  "note": "Security lab that evaluates frontier models for offensive cyber capabilities on behalf of AI developers."
 },
 {
  "id": "606b196446",
  "title": "It Sounds Crazy, But Is It? Peter Doocy Presses Karine Jean-Pierre On AI Becoming Self-Aware",
  "creators": "Peter Doocy; Karine Jean-Pierre",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Forbes Breaking News (White House briefing)",
  "url": "https://www.youtube.com/watch?v=dSmJuVsgIE8",
  "note": "A White House press briefing exchange prompted by Yudkowsky's warnings about AI."
 },
 {
  "id": "02511f77e5",
  "title": "Jailbroken: How Does LLM Safety Training Fail?",
  "creators": "Alexander Wei, Nika Haghtalab, Jacob Steinhardt",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2307.02483",
  "note": "Attributes jailbreaks to competing objectives and mismatched generalisation."
 },
 {
  "id": "edc094311d",
  "title": "Journalist had a creepy encounter with new tech that left him unable to sleep",
  "creators": "Kevin Roose",
  "year": 2023,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "CNN",
  "url": "https://www.youtube.com/watch?v=f24JL0nnhcA",
  "note": "Kevin Roose describes his conversation with Bing's Sydney persona."
 },
 {
  "id": "f3ce7cc96a",
  "title": "Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena",
  "creators": "Lianmin Zheng et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.05685",
  "note": "Introduced MT-Bench and validated strong models as judges of chat quality."
 },
 {
  "id": "13aa0aa5d7",
  "title": "LISA (London Initiative for Safe AI)",
  "creators": "London, UK",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.safeai.org.uk",
  "note": "London coworking and research hub for AI safety organizations and researchers."
 },
 {
  "id": "d3b7083053",
  "title": "LLM Guard",
  "creators": "Protect AI",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/protectai/llm-guard",
  "note": "Security toolkit for sanitizing LLM inputs and outputs."
 },
 {
  "id": "2dd6c1a3b2",
  "title": "LMArena",
  "creators": "LMArena (formerly LMSYS Chatbot Arena)",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "LMArena",
  "url": "https://arena.ai/",
  "note": "Crowdsourced leaderboard of models ranked by pairwise human preference."
 },
 {
  "id": "b04476efb5",
  "title": "LMSYS-Chat-1M",
  "creators": "LMSYS",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2309.11998",
  "note": "One million real world conversations with 25 language models."
 },
 {
  "id": "cf7501089f",
  "title": "Language Models Can Explain Neurons in Language Models",
  "creators": "Steven Bills et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "OpenAI",
  "url": "https://openaipublic.blob.core.windows.net/neuron-explainer/paper/index.html",
  "note": "Uses GPT-4 to write and score explanations of GPT-2 neurons."
 },
 {
  "id": "bd1abaa9dc",
  "title": "Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting",
  "creators": "Miles Turpin, Julian Michael, Ethan Perez, Samuel R. Bowman",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2305.04388",
  "note": "Shows chain-of-thought explanations can systematically misrepresent the true reason for a prediction."
 },
 {
  "id": "2ee7bc57b7",
  "title": "Language Models Represent Space and Time",
  "creators": "Wes Gurnee, Max Tegmark",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2310.02207",
  "note": "Shows LLMs learn linear representations of geography and time."
 },
 {
  "id": "d42b688149",
  "title": "Language models surprised us",
  "creators": "Ajeya Cotra",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Planned Obsolescence",
  "url": "https://www.planned-obsolescence.org/language-models-surprised-us/",
  "note": "Notes that forecasters and experts underestimated LLM progress."
 },
 {
  "id": "9519be3e58",
  "title": "Large Language Models can Strategically Deceive their Users when Put Under Pressure",
  "creators": "Jérémy Scheurer, Mikita Balesni, Marius Hobbhahn",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.07590",
  "note": "Shows GPT-4 acting as a trading agent performs insider trading and then lies about it."
 },
 {
  "id": "cd08af1693",
  "title": "Lennart Heim on Compute Governance",
  "creators": "Lennart Heim; Gus Docker",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=iCxJUDDvq94",
  "note": "Heim explains why compute is a useful lever for AI governance."
 },
 {
  "id": "4d2b169941",
  "title": "Levels of AGI for Operationalizing Progress on the Path to AGI",
  "creators": "Meredith Ringel Morris et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Forecasting"
  ],
  "publisher": "ICML 2024",
  "url": "https://arxiv.org/abs/2311.02462",
  "note": "Proposes a framework for classifying AGI by performance and generality."
 },
 {
  "id": "c56680a36f",
  "title": "Llama 2: Open Foundation and Fine-Tuned Chat Models",
  "creators": "Meta",
  "year": 2023,
  "kind": "System card",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.09288",
  "note": "Detailed safety fine tuning, red teaming, and evaluation of Llama 2 Chat."
 },
 {
  "id": "f08afb0844",
  "title": "Llama Guard",
  "creators": "Meta",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.06674",
  "note": "LLM based input and output safeguard classifier for human AI conversations."
 },
 {
  "id": "9290d2ffce",
  "title": "METR Blog",
  "creators": "METR",
  "year": 2023,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "METR",
  "url": "https://metr.org/blog/",
  "note": "Research updates on evaluating autonomous capabilities of frontier models."
 },
 {
  "id": "06943b4fbf",
  "title": "MLAgentBench",
  "creators": "Qian Huang et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.03302",
  "note": "Suite of ML experimentation tasks for evaluating research agents."
 },
 {
  "id": "d1b1066942",
  "title": "MLST Live: George Hotz and Connor Leahy on AI Safety",
  "creators": "George Hotz; Connor Leahy",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=8LxTWIaInok",
  "note": "A live debate on AI risk."
 },
 {
  "id": "beddb01886",
  "title": "Managing Extreme AI Risks amid Rapid Progress",
  "creators": "Yoshua Bengio, Geoffrey Hinton et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Science (2024)",
  "url": "https://arxiv.org/abs/2310.17688",
  "note": "Leading researchers call for technical R&D and adaptive governance to address extreme AI risks."
 },
 {
  "id": "91e60e8dbe",
  "title": "Map of AI Existential Safety",
  "creators": "AISafety.com",
  "year": 2023,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AISafety.com",
  "url": "https://www.aisafety.com/map",
  "note": "An illustrated map of organizations working on AI existential safety."
 },
 {
  "id": "4fac9d1747",
  "title": "Max Tegmark: The Case for Halting AI Development, Lex Fridman Podcast #371",
  "creators": "Max Tegmark; Lex Fridman",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=VcVfceTsD0A",
  "note": "Tegmark explains the Future of Life Institute open letter calling for a pause on giant AI experiments."
 },
 {
  "id": "c1116a1241",
  "title": "Measuring Faithfulness in Chain-of-Thought Reasoning",
  "creators": "Tamera Lanham et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.13702",
  "note": "Tests how much models actually rely on their stated reasoning."
 },
 {
  "id": "3899e5e3c8",
  "title": "Mechanistic Interpretability: Neel Nanda (DeepMind)",
  "creators": "Neel Nanda; Tim Scarfe",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=_Ygf0GnlwmY",
  "note": "A four hour technical discussion of mechanistic interpretability."
 },
 {
  "id": "5e721027ab",
  "title": "Model Evaluation for Extreme Risks",
  "creators": "Toby Shevlane et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2305.15324",
  "note": "Argues dangerous capability and alignment evaluations are critical for frontier AI governance."
 },
 {
  "id": "4b8bf8310b",
  "title": "More than a Glitch: Confronting Race, Gender, and Ability Bias in Tech",
  "creators": "Meredith Broussard",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=More+than+a+Glitch+Broussard",
  "note": "Argues that algorithmic bias is structural rather than a fixable bug."
 },
 {
  "id": "fc48797eed",
  "title": "Munk Debate on Artificial Intelligence: Bengio and Tegmark vs. Mitchell and LeCun",
  "creators": "Yoshua Bengio; Max Tegmark; Melanie Mitchell; Yann LeCun",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Munk Debates",
  "url": "https://www.youtube.com/watch?v=144uOfr4SYA",
  "note": "A formal debate on whether AI research poses an existential threat."
 },
 {
  "id": "d4ac9adde8",
  "title": "My techno-optimism",
  "creators": "Vitalik Buterin",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "vitalik.eth.limo",
  "url": "https://vitalik.eth.limo/general/2023/11/27/techno_optimism.html",
  "note": "Introduces defensive accelerationism, or d/acc, as a middle path on AI risk."
 },
 {
  "id": "4b21253fe5",
  "title": "NCSL Artificial Intelligence Legislation Database",
  "creators": "National Conference of State Legislatures",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Policy"
  ],
  "publisher": "NCSL",
  "url": "https://www.ncsl.org/technology-and-communication/artificial-intelligence-2025-legislation",
  "note": "Summary of AI bills introduced in US state legislatures."
 },
 {
  "id": "d154e67a4b",
  "title": "NIST AI Risk Management Framework (AI RMF 1.0)",
  "creators": "US National Institute of Standards and Technology",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://www.nist.gov/itl/ai-risk-management-framework",
  "note": "Voluntary framework organized around govern, map, measure, and manage functions for AI risk."
 },
 {
  "id": "6ab8ea4e6e",
  "title": "Natural Selection Favors AIs over Humans",
  "creators": "Dan Hendrycks",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2303.16200",
  "note": "Argues competitive pressures will select for selfish AI agents."
 },
 {
  "id": "80a49df214",
  "title": "NeMo Guardrails",
  "creators": "NVIDIA",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/NVIDIA-NeMo/Guardrails",
  "note": "Toolkit for adding programmable guardrails to LLM applications."
 },
 {
  "id": "1eb96c4311",
  "title": "Neuronpedia",
  "creators": "Johnny Lin and Joseph Bloom",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Neuronpedia",
  "url": "https://www.neuronpedia.org/",
  "note": "Open platform for exploring sparse autoencoder features, attribution graphs, and steering."
 },
 {
  "id": "a3878688f2",
  "title": "Nobody's on the ball on AGI alignment",
  "creators": "Leopold Aschenbrenner",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "For Our Posterity",
  "url": "https://www.forourposterity.com/nobodys-on-the-ball-on-agi-alignment/",
  "note": "Argues far too few serious researchers are working on aligning superhuman AI."
 },
 {
  "id": "27940d2e87",
  "title": "Not What You've Signed Up For: Compromising Real-World LLM-Integrated Applications with Indirect Prompt Injection",
  "creators": "Kai Greshake et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "AISec",
  "url": "https://arxiv.org/abs/2302.12173",
  "note": "Introduces indirect prompt injection via retrieved content."
 },
 {
  "id": "b5fc77236f",
  "title": "OECD AI Incidents Monitor (AIM)",
  "creators": "OECD",
  "year": 2023,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://oecd.ai/en/incidents",
  "note": "OECD monitor tracking AI incidents reported in global news."
 },
 {
  "id": "b5412b76f3",
  "title": "OWASP Top 10 for LLM Applications",
  "creators": "OWASP GenAI Security Project",
  "year": 2023,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "OWASP",
  "url": "https://genai.owasp.org/",
  "note": "Community ranked list of the most critical security risks for LLM applications."
 },
 {
  "id": "a8988deeb3",
  "title": "Open LLM Leaderboard",
  "creators": "Hugging Face",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Hugging Face",
  "url": "https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard",
  "note": "Leaderboard of open weights models evaluated with a standard harness."
 },
 {
  "id": "1f0388c993",
  "title": "Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback",
  "creators": "Stephen Casper et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2307.15217",
  "note": "Surveys the limitations of RLHF and argues it is not sufficient for safe AI."
 },
 {
  "id": "44d7171752",
  "title": "Open-Sourcing Highly Capable Foundation Models",
  "creators": "Elizabeth Seger et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.09227",
  "note": "Weighs the benefits and risks of releasing model weights and proposes alternatives."
 },
 {
  "id": "b110bba97a",
  "title": "OpenAI CEO Sam Altman testifies at Senate artificial intelligence hearing (full video)",
  "creators": "Sam Altman; Gary Marcus; Christina Montgomery",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "CBS News (U.S. Senate Judiciary Subcommittee)",
  "url": "https://www.youtube.com/watch?v=TO0J2Yw7usM",
  "note": "Full video of the May 2023 Senate Judiciary subcommittee hearing on rules for artificial intelligence."
 },
 {
  "id": "a5727ddf0c",
  "title": "OpenAI Evals",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/openai/evals",
  "note": "Framework and open registry of benchmarks for evaluating language models."
 },
 {
  "id": "6141df7e4e",
  "title": "OpenAI Preparedness Framework",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://openai.com/index/updating-our-preparedness-framework/",
  "note": "Beta released December 2023 and version 2 published April 2025, tracking biological, cyber, and self-improvement capabilities."
 },
 {
  "id": "8142cf14bc",
  "title": "OpenAI's huge push to make superintelligence safe",
  "creators": "Jan Leike; Rob Wiblin",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=ZP_N4q5U3eE",
  "note": "Leike explains the goals and plan of OpenAI's Superalignment team."
 },
 {
  "id": "c731eebed2",
  "title": "OpenCompass",
  "creators": "Shanghai AI Laboratory",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/open-compass/opencompass",
  "note": "Evaluation platform supporting a wide range of models and benchmarks."
 },
 {
  "id": "0b6af9b8c0",
  "title": "Oversight for Frontier AI through a Know-Your-Customer Scheme for Compute Providers",
  "creators": "Janet Egan, Lennart Heim",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.13625",
  "note": "Proposes KYC requirements for cloud compute providers."
 },
 {
  "id": "9c83cb1abe",
  "title": "Oversight of A.I.: Principles for Regulation",
  "creators": "Dario Amodei; Yoshua Bengio; Stuart Russell",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy",
   "Biosecurity"
  ],
  "publisher": "U.S. Senate Judiciary Subcommittee on Privacy, Technology, and the Law",
  "url": "https://www.judiciary.senate.gov/committee-activity/hearings/oversight-of-ai-principles-for-regulation",
  "note": "Official hearing page for the July 25, 2023 session with three leading voices on AI risk."
 },
 {
  "id": "a07bf3af40",
  "title": "Oversight of A.I.: Rules for Artificial Intelligence",
  "creators": "Sam Altman; Gary Marcus; Christina Montgomery",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "U.S. Senate Judiciary Subcommittee on Privacy, Technology, and the Law",
  "url": "https://www.judiciary.senate.gov/committee-activity/hearings/oversight-of-ai-rules-for-artificial-intelligence",
  "note": "Official hearing page with video and written testimony from Altman's first appearance before Congress on May 16, 2023."
 },
 {
  "id": "1a43f21bc8",
  "title": "Owain Evans: Truthful language models and AI alignment",
  "creators": "Owain Evans",
  "year": 2023,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Schwartz Reisman Institute",
  "url": "https://www.youtube.com/watch?v=AkLkZgsaKp4",
  "note": "Evans discusses TruthfulQA and the goal of honest AI."
 },
 {
  "id": "e9aa935a8c",
  "title": "Oxford Martin AI Governance Initiative",
  "creators": "Oxford, UK",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Academic",
  "url": "https://aigi.ox.ac.uk",
  "note": "University of Oxford initiative researching technical and international AI governance."
 },
 {
  "id": "538563f304",
  "title": "PAI Guidance for Safe Foundation Model Deployment",
  "creators": "Partnership on AI",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://partnershiponai.org/modeldeployment/",
  "note": "Framework of recommended practices tailored to model capability and release type."
 },
 {
  "id": "c7d275030c",
  "title": "Palisade Research",
  "creators": "Berkeley, CA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://palisaderesearch.org",
  "note": "Studies offensive and dangerous AI capabilities, including experiments on shutdown resistance in frontier models."
 },
 {
  "id": "8fdf3fdd2c",
  "title": "Paul Christiano: Preventing an AI takeover",
  "creators": "Paul Christiano; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=9AAhTLa0dT0",
  "note": "Christiano discusses responsible scaling policies and his research on heuristic arguments."
 },
 {
  "id": "f03a9b5ebb",
  "title": "Pause Giant AI Experiments: An Open Letter",
  "creators": "Future of Life Institute",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org/open-letter/pause-giant-ai-experiments/",
  "note": "March 2023 letter calling for a six-month pause on training systems more powerful than GPT-4."
 },
 {
  "id": "3eeb4c9e11",
  "title": "PauseAI",
  "creators": "International",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://pauseai.info",
  "note": "Grassroots movement campaigning for an international pause on frontier AI development."
 },
 {
  "id": "d154fa7ff2",
  "title": "Pausing AI Developments Isn't Enough. We Need to Shut it All Down",
  "creators": "Eliezer Yudkowsky",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "TIME",
  "url": "https://time.com/6266923/ai-eliezer-yudkowsky-open-letter-not-enough/",
  "note": "Calls for an indefinite worldwide moratorium on large training runs, enforced by international agreement."
 },
 {
  "id": "0e0a320e9a",
  "title": "Planned Obsolescence",
  "creators": "Ajeya Cotra, Kelsey Piper",
  "year": 2023,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Planned Obsolescence",
  "url": "https://www.planned-obsolescence.org/",
  "note": "A blog about AI futurism and alignment by Ajeya Cotra and Kelsey Piper."
 },
 {
  "id": "4ba2ae8a3f",
  "title": "Planning for AGI and beyond",
  "creators": "Sam Altman",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/planning-for-agi-and-beyond/",
  "note": "OpenAI's statement of principles for the transition to AGI."
 },
 {
  "id": "413971bd20",
  "title": "Poisoning Web-Scale Training Datasets is Practical",
  "creators": "Nicholas Carlini et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "IEEE S&P 2024",
  "url": "https://arxiv.org/abs/2302.10149",
  "note": "Demonstrates cheap poisoning attacks on web-scale datasets."
 },
 {
  "id": "306d5f43f3",
  "title": "Political Declaration on Responsible Military Use of AI and Autonomy",
  "creators": "US Department of State",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "International",
  "url": "https://www.state.gov/political-declaration-on-responsible-military-use-of-artificial-intelligence-and-autonomy-2/",
  "note": "Declaration launched in 2023 and endorsed by dozens of states on military AI norms."
 },
 {
  "id": "e202c1b4de",
  "title": "Practices for Governing Agentic AI Systems",
  "creators": "Yonadav Shavit et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Agents",
   "Governance"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/practices-for-governing-agentic-ai-systems/",
  "note": "Proposes baseline responsibilities for parties deploying agentic AI."
 },
 {
  "id": "dd28569d8c",
  "title": "Pretraining Language Models with Human Preferences",
  "creators": "Tomasz Korbak et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2302.08582",
  "note": "Finds conditional training on human preference signals during pretraining reduces undesirable outputs."
 },
 {
  "id": "fd53e1a7f3",
  "title": "Progress Measures for Grokking via Mechanistic Interpretability",
  "creators": "Neel Nanda, Lawrence Chan, Tom Lieberum, Jess Smith, Jacob Steinhardt",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICLR",
  "url": "https://arxiv.org/abs/2301.05217",
  "note": "Fully reverse-engineers the algorithm a small transformer learns for modular addition."
 },
 {
  "id": "494a17f6e2",
  "title": "PromptBench",
  "creators": "Kaijie Zhu et al.",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.07910",
  "note": "Unified library for evaluating language models including adversarial prompt robustness."
 },
 {
  "id": "7d7ef16715",
  "title": "Purple Llama",
  "creators": "Meta",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/meta-llama/PurpleLlama",
  "note": "Umbrella project for Llama Guard, Prompt Guard, and CyberSecEval."
 },
 {
  "id": "1629ab52d5",
  "title": "Purple Llama CyberSecEval",
  "creators": "Meta",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2312.04724",
  "note": "Benchmark for insecure code generation and compliance with cyberattack requests."
 },
 {
  "id": "4b4cfc11a3",
  "title": "Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling",
  "creators": "EleutherAI",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2304.01373",
  "note": "Sixteen models with public checkpoints and data order for interpretability research."
 },
 {
  "id": "db2f682e68",
  "title": "RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback",
  "creators": "Harrison Lee et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2309.00267",
  "note": "Shows AI-generated preference labels can match human labels for RL fine-tuning."
 },
 {
  "id": "1a7ee24ba8",
  "title": "Representation Engineering: A Top-Down Approach to AI Transparency",
  "creators": "Andy Zou et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.01405",
  "note": "Reads and controls high-level concepts like honesty through population-level representations."
 },
 {
  "id": "d89dba9af5",
  "title": "Responsible Scaling Policies (METR overview)",
  "creators": "METR",
  "year": 2023,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://metr.org/blog/2023-09-26-rsp/",
  "note": "Essay introducing the concept of responsible scaling policies for AI developers."
 },
 {
  "id": "e877de6016",
  "title": "Rishi Sunak and Elon Musk: Talk AI, Tech and the Future",
  "creators": "Rishi Sunak; Elon Musk",
  "year": 2023,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Rishi Sunak (YouTube)",
  "url": "https://www.youtube.com/watch?v=R2meHtrO1n8",
  "note": "A conversation held at the close of the 2023 Bletchley Park AI Safety Summit."
 },
 {
  "id": "97e61b2142",
  "title": "SPAR: Supervised Program for Alignment Research",
  "creators": "SPAR",
  "year": 2023,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "SPAR",
  "url": "https://sparai.org/",
  "note": "A part-time remote program pairing mentees with AI safety mentors."
 },
 {
  "id": "dafd19e1fb",
  "title": "SWE-bench: Can Language Models Resolve Real-World GitHub Issues?",
  "creators": "Carlos E. Jimenez et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2310.06770",
  "note": "A widely used benchmark of real software engineering tasks."
 },
 {
  "id": "6bcc31ad4f",
  "title": "Safe AI Forum (SAIF)",
  "creators": "International",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://saif.org",
  "note": "Nonprofit that organizes the International Dialogues on AI Safety between Western and Chinese scientists."
 },
 {
  "id": "2950aeb5c4",
  "title": "Safety Testing Chair's Statement (Bletchley)",
  "creators": "UK Government",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Evaluations"
  ],
  "publisher": "United Kingdom",
  "url": "https://www.gov.uk/government/publications/ai-safety-summit-2023-chairs-statement-safety-testing-2-november",
  "note": "Statement agreeing that governments would collaborate on testing frontier models before deployment."
 },
 {
  "id": "4e08d114e0",
  "title": "Safety-Gymnasium",
  "creators": "PKU Alignment",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.12567",
  "note": "Unified safe reinforcement learning benchmark suite."
 },
 {
  "id": "a4035d5d8a",
  "title": "Sam Altman: OpenAI CEO on GPT-4, ChatGPT, and the Future of AI, Lex Fridman Podcast #367",
  "creators": "Sam Altman; Lex Fridman",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=L_Guz73e6fw",
  "note": "Altman discusses GPT-4, alignment and AI safety shortly after its launch."
 },
 {
  "id": "6954d73592",
  "title": "Scale AI Safety, Evaluations and Analysis Lab (SEAL)",
  "creators": "San Francisco, CA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Company",
  "url": "https://scale.com/research",
  "note": "Research lab at Scale AI publishing benchmarks and red teaming research."
 },
 {
  "id": "d702127e57",
  "title": "Scheming AIs: Will AIs Fake Alignment During Training in Order to Get Power?",
  "creators": "Joe Carlsmith",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.08379",
  "note": "Assesses how likely training is to produce models that strategically fake alignment."
 },
 {
  "id": "f4ffa0139f",
  "title": "Science, Innovation and Technology Committee's inquiry on the governance of AI",
  "creators": "UK House of Commons Science, Innovation and Technology Committee",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Sky News (UK Parliament)",
  "url": "https://www.youtube.com/watch?v=XJnwf-cVIqQ",
  "note": "An oral evidence session in the UK Parliament's inquiry into the governance of artificial intelligence."
 },
 {
  "id": "a5b4605716",
  "title": "Science, Innovation and Technology Committee's inquiry on the governance of AI (second session)",
  "creators": "UK House of Commons Science, Innovation and Technology Committee",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Sky News (UK Parliament)",
  "url": "https://www.youtube.com/watch?v=lq6oEVbC-tE",
  "note": "A further evidence session in the Commons committee's governance of AI inquiry."
 },
 {
  "id": "f231177b65",
  "title": "Scott Aaronson Talks AI Safety",
  "creators": "Scott Aaronson",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Effective Altruism at UT Austin",
  "url": "https://www.youtube.com/watch?v=fc-cHk9yFpg",
  "note": "Aaronson reflects on AI safety after his work at OpenAI."
 },
 {
  "id": "9453274412",
  "title": "Senate Judiciary Committee holds hearing on AI oversight and regulation, 07/25/23",
  "creators": "Dario Amodei; Yoshua Bengio; Stuart Russell",
  "year": 2023,
  "kind": "Testimony",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy",
   "Biosecurity"
  ],
  "publisher": "CNBC Television (U.S. Senate Judiciary Subcommittee)",
  "url": "https://www.youtube.com/watch?v=hm1zexCjELo",
  "note": "The July 2023 hearing on principles for AI regulation where Amodei warned about AI enabled biological attacks."
 },
 {
  "id": "b5f462dc82",
  "title": "Shane Legg (DeepMind Founder): 2028 AGI, superhuman alignment, new architectures",
  "creators": "Shane Legg; Dwarkesh Patel",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.youtube.com/watch?v=Kc1atfJkiJU",
  "note": "Legg discusses his longstanding AGI timeline forecast and approaches to aligning superhuman systems."
 },
 {
  "id": "39dfffe29f",
  "title": "Simple Synthetic Data Reduces Sycophancy in Large Language Models",
  "creators": "Jerry Wei, Da Huang, Yifeng Lu, Denny Zhou, Quoc V. Le",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.03958",
  "note": "Reduces sycophancy with a lightweight synthetic data fine-tuning step."
 },
 {
  "id": "441c3460f2",
  "title": "SimpleSafetyTests",
  "creators": "Bertie Vidgen et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.08370",
  "note": "100 clear cut harmful prompts for quickly identifying critical safety risks."
 },
 {
  "id": "60ae408fda",
  "title": "Sociotechnical Safety Evaluation of Generative AI Systems",
  "creators": "Laura Weidinger et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.11986",
  "note": "Proposes evaluating capability, human interaction and systemic impact layers."
 },
 {
  "id": "e540f27e44",
  "title": "Sparse Autoencoders Find Highly Interpretable Features in Language Models",
  "creators": "Hoagy Cunningham, Aidan Ewart, Logan Riggs, Robert Huben, Lee Sharkey",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2309.08600",
  "note": "Shows sparse dictionary learning resolves superposition into interpretable features."
 },
 {
  "id": "b2d52be78d",
  "title": "Specific versus General Principles for Constitutional AI",
  "creators": "Sandipan Kundu et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.13798",
  "note": "Tests whether a single general principle can replace detailed constitutions."
 },
 {
  "id": "f0b655b8e4",
  "title": "Statement on AI Risk",
  "creators": "Center for AI Safety",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://safe.ai/work/statement-on-ai-risk",
  "note": "One-sentence statement that mitigating the risk of extinction from AI should be a global priority, signed by leading researchers."
 },
 {
  "id": "72b124f60c",
  "title": "Steering Language Models With Activation Engineering",
  "creators": "Alexander Matt Turner et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.10248",
  "note": "Introduces activation addition to steer model outputs at inference time."
 },
 {
  "id": "eea41de4a9",
  "title": "Steering Llama 2 via Contrastive Activation Addition",
  "creators": "Nina Rimsky, Nick Gabrieli, Julian Schulz, Meg Tong, Evan Hubinger, Alexander Matt Turner",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "ACL 2024",
  "url": "https://arxiv.org/abs/2312.06681",
  "note": "Steers behaviours like sycophancy using contrastive activation vectors."
 },
 {
  "id": "2e642e6c57",
  "title": "Studying Large Language Model Generalization with Influence Functions",
  "creators": "Roger Grosse et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.03296",
  "note": "Scales influence functions to trace model behaviour back to training data."
 },
 {
  "id": "1b5a57e4d4",
  "title": "Taken Out of Context: On Measuring Situational Awareness in LLMs",
  "creators": "Lukas Berglund et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2309.00667",
  "note": "Studies out-of-context reasoning as a precursor to situational awareness."
 },
 {
  "id": "1b54c02ce3",
  "title": "Tensor Trust",
  "creators": "Sam Toyer et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2311.01011",
  "note": "Human generated prompt injection attacks and defenses collected from an online game."
 },
 {
  "id": "14b408d0aa",
  "title": "The 'Don't Look Up' Thinking That Could Doom Us With AI",
  "creators": "Max Tegmark",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "TIME",
  "url": "https://time.com/6273743/thinking-that-could-doom-us-with-ai/",
  "note": "Compares complacency about AI risk to the film Don't Look Up."
 },
 {
  "id": "d5553ed2bc",
  "title": "The A.I. Dilemma, March 9, 2023",
  "creators": "Tristan Harris; Aza Raskin",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Center for Humane Technology",
  "url": "https://www.youtube.com/watch?v=xoVJKj8lcNQ",
  "note": "The Center for Humane Technology presentation on the risks of racing to deploy large language models."
 },
 {
  "id": "54bd139e13",
  "title": "The AI Alignment Debate: Can We Develop Truly Beneficial AI?",
  "creators": "George Hotz; Connor Leahy",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Machine Learning Street Talk",
  "url": "https://www.youtube.com/watch?v=iFUmWho7fBE",
  "note": "A higher quality version of the Hotz and Leahy debate on alignment."
 },
 {
  "id": "ff50c3c0d1",
  "title": "The AI Dilemma (Your Undivided Attention episode)",
  "creators": "Tristan Harris; Aza Raskin",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Center for Humane Technology",
  "url": "https://www.humanetech.com/podcast/the-ai-dilemma",
  "note": "The audio version of the AI Dilemma presentation on the Your Undivided Attention podcast."
 },
 {
  "id": "852b779e66",
  "title": "The AI Power Paradox",
  "creators": "Ian Bremmer, Mustafa Suleyman",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Foreign Affairs",
  "url": "https://www.foreignaffairs.com/world/artificial-intelligence-power-paradox",
  "note": "Proposes technoprudential governance of AI by states and companies."
 },
 {
  "id": "f8384c854e",
  "title": "The AI Safety Summit",
  "creators": "UK Department for Science, Innovation and Technology",
  "year": 2023,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "UK Department for Science, Innovation and Technology",
  "url": "https://www.youtube.com/watch?v=wFBmMlig1tE",
  "note": "The UK government's video for the November 2023 AI Safety Summit."
 },
 {
  "id": "c8f1209722",
  "title": "The Capacity for Moral Self-Correction in Large Language Models",
  "creators": "Deep Ganguli et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2302.07459",
  "note": "Shows large RLHF models can reduce biased outputs when instructed to."
 },
 {
  "id": "c29c906c32",
  "title": "The Cognitive Revolution",
  "creators": "Nathan Labenz",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Cognitive Revolution",
  "url": "https://www.cognitiverevolution.ai/",
  "note": "An AI podcast with frequent interviews with safety and interpretability researchers."
 },
 {
  "id": "bc705e373a",
  "title": "The Coming Wave: Technology, Power, and the Twenty-First Century's Greatest Dilemma",
  "creators": "Mustafa Suleyman, Michael Bhaskar",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Biosecurity"
  ],
  "publisher": "Crown",
  "url": "https://openlibrary.org/search?q=The+Coming+Wave+Suleyman",
  "note": "Argues that AI and synthetic biology must be contained through a mix of technical, corporate and state measures."
 },
 {
  "id": "c6cb6056ba",
  "title": "The Ethics of Artificial Intelligence: Principles, Challenges, and Opportunities",
  "creators": "Luciano Floridi",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=The+Ethics+of+Artificial+Intelligence+Floridi",
  "note": "A philosophical framework for AI ethics and its translation into governance."
 },
 {
  "id": "aa2684c13a",
  "title": "The Exciting, Perilous Journey Toward AGI",
  "creators": "Ilya Sutskever",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=SEkGLj0bwAU",
  "note": "Sutskever's TED talk on why AGI will be transformative and why safety will become a shared concern."
 },
 {
  "id": "29b3893e0d",
  "title": "The Foundation Model Transparency Index",
  "creators": "Rishi Bommasani et al.",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.12941",
  "note": "Paper introducing the first edition of the index."
 },
 {
  "id": "f83ea2f58b",
  "title": "The Geometry of Truth: Emergent Linear Structure in Large Language Model Representations of True/False Datasets",
  "creators": "Samuel Marks, Max Tegmark",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2310.06824",
  "note": "Finds linear representations of truth in LLM activations."
 },
 {
  "id": "f29fdb682f",
  "title": "The Path to AI Arms Control",
  "creators": "Henry Kissinger, Graham Allison",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Foreign Affairs",
  "url": "https://www.foreignaffairs.com/united-states/henry-kissinger-path-artificial-intelligence-arms-control",
  "note": "Draws on nuclear arms control history to propose US-China AI risk dialogue."
 },
 {
  "id": "552a9806cb",
  "title": "The Power of Intelligence: An Essay By Eliezer Yudkowsky",
  "creators": "Rational Animations",
  "year": 2023,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Rational Animations",
  "url": "https://www.youtube.com/watch?v=q9Figerh89g",
  "note": "An animated reading of Yudkowsky's essay on why intelligence is a powerful force."
 },
 {
  "id": "59dcf7fc13",
  "title": "The Urgent Risks of Runaway AI and What to Do about Them",
  "creators": "Gary Marcus",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=JL5OFXeXenA",
  "note": "Marcus calls for a global, neutral AI governance agency."
 },
 {
  "id": "dd571259e6",
  "title": "The Waluigi Effect (mega-post)",
  "creators": "Cleo Nardo",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/D7PumeYTDPfBTp3i7/the-waluigi-effect-mega-post",
  "note": "Argues training a model to have a property makes its opposite easy to elicit."
 },
 {
  "id": "d2fce3e68f",
  "title": "The Worlds I See: Curiosity, Exploration, and Discovery at the Dawn of AI",
  "creators": "Fei-Fei Li",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "History"
  ],
  "publisher": "Flatiron Books",
  "url": "https://openlibrary.org/search?q=The+Worlds+I+See+Fei-Fei+Li",
  "note": "A memoir of the rise of modern computer vision and a call for human-centred AI."
 },
 {
  "id": "f536c9109c",
  "title": "The final push for AGI, OpenAI's leadership drama, and red-teaming frontier models",
  "creators": "Nathan Labenz; Rob Wiblin",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=_APDe0Mnen4",
  "note": "Labenz recounts red-teaming GPT-4 before release."
 },
 {
  "id": "4fc374377b",
  "title": "This Changes Everything",
  "creators": "Ezra Klein",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "The New York Times",
  "url": "https://www.nytimes.com/2023/03/12/opinion/chatbots-artificial-intelligence-future-weirdness.html",
  "note": "An opinion column on the speed and strangeness of AI progress."
 },
 {
  "id": "ca697df7bc",
  "title": "Timaeus",
  "creators": "Berkeley, CA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "Nonprofit",
  "url": "https://timaeus.co",
  "note": "Research organization applying singular learning theory to developmental interpretability."
 },
 {
  "id": "fe0da16d3d",
  "title": "ToolEmu: Identifying the Risks of LM Agents with an LM-Emulated Sandbox",
  "creators": "Yangjun Ruan et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2309.15817",
  "note": "Uses an LM to emulate tools and surface risky agent actions at scale."
 },
 {
  "id": "34f1bdb157",
  "title": "Towards Automated Circuit Discovery for Mechanistic Interpretability",
  "creators": "Arthur Conmy, Augustine N. Mavor-Parker, Aengus Lynch, Stefan Heimersheim, Adrià Garriga-Alonso",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2304.14997",
  "note": "Automates the search for circuits responsible for a behaviour."
 },
 {
  "id": "12258e34f6",
  "title": "Towards Best Practices in AGI Safety and Governance: A Survey of Expert Opinion",
  "creators": "Jonas Schuett et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2305.07153",
  "note": "Surveys experts, who broadly endorse practices like pre-deployment risk assessment."
 },
 {
  "id": "97b521f0a6",
  "title": "Towards Measuring the Representation of Subjective Global Opinions in Language Models",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2306.16388",
  "note": "GlobalOpinionQA dataset comparing model answers to cross national survey responses."
 },
 {
  "id": "cbf2b26867",
  "title": "Towards Monosemanticity",
  "creators": "Anthropic",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits",
  "url": "https://transformer-circuits.pub/2023/monosemantic-features",
  "note": "Dictionary learning study with an interactive feature browser."
 },
 {
  "id": "3c61fc4960",
  "title": "Towards Monosemanticity: Decomposing Language Models With Dictionary Learning",
  "creators": "Trenton Bricken et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2023/monosemantic-features/index.html",
  "note": "Uses sparse autoencoders to extract interpretable features from a one-layer transformer."
 },
 {
  "id": "f819fb0e43",
  "title": "Towards Understanding Sycophancy in Language Models",
  "creators": "Mrinank Sharma et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2310.13548",
  "note": "Shows RLHF assistants consistently tell users what they want to hear."
 },
 {
  "id": "8217211fc8",
  "title": "Tracr: Compiled Transformers as a Laboratory for Interpretability",
  "creators": "Google DeepMind",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2301.05062",
  "note": "Compiles programs into transformer weights to provide ground truth circuits."
 },
 {
  "id": "e3815e5cca",
  "title": "Trends in Artificial Intelligence",
  "creators": "Epoch AI",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/trends",
  "note": "Dashboard of key trends in compute, data, hardware, and algorithmic progress."
 },
 {
  "id": "9eac8d3e76",
  "title": "UK AI Security Institute",
  "creators": "London, UK",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance",
   "Cybersecurity"
  ],
  "publisher": "Government",
  "url": "https://www.aisi.gov.uk",
  "note": "Launched as the UK AI Safety Institute after the Bletchley summit and renamed the AI Security Institute in February 2025."
 },
 {
  "id": "0a0a109328",
  "title": "UNKNOWN: Killer Robots (official trailer)",
  "creators": "Jesse Sweet",
  "year": 2023,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Biosecurity"
  ],
  "publisher": "Netflix",
  "url": "https://www.youtube.com/watch?v=YsSzNOpr9cE",
  "note": "Trailer for the Netflix documentary on military AI and AI-enabled drug discovery misuse."
 },
 {
  "id": "cac24b7ffe",
  "title": "US Center for AI Standards and Innovation (CAISI)",
  "creators": "Gaithersburg, MD",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Policy"
  ],
  "publisher": "Government",
  "url": "https://www.nist.gov/caisi",
  "note": "Created within NIST as the US AI Safety Institute and reorganized as CAISI in June 2025."
 },
 {
  "id": "cd676102c7",
  "title": "US State AI Governance Legislation Tracker",
  "creators": "IAPP",
  "year": 2023,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Policy"
  ],
  "publisher": "IAPP",
  "url": "https://iapp.org/resources/article/us-state-ai-governance-legislation-tracker/",
  "note": "Tracker of cross sector AI governance bills in US states."
 },
 {
  "id": "5a208a2ce1",
  "title": "Uncontrollable: The Threat of Artificial Superintelligence and the Race to Save the World",
  "creators": "Darren McKee",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Self-published",
  "url": "https://openlibrary.org/search?q=Uncontrollable+Darren+McKee",
  "note": "An accessible introduction to the risks of superintelligent AI and what individuals and governments can do about them."
 },
 {
  "id": "8fff7576d2",
  "title": "Understanding AI",
  "creators": "Timothy B. Lee",
  "year": 2023,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Policy",
   "Forecasting"
  ],
  "publisher": "Substack",
  "url": "https://www.understandingai.org/",
  "note": "Explainers and reporting on AI capabilities and policy."
 },
 {
  "id": "38666c1278",
  "title": "Universal and Transferable Adversarial Attacks on Aligned Language Models",
  "creators": "Andy Zou, Zifan Wang, Nicholas Carlini, Milad Nasr, J. Zico Kolter, Matt Fredrikson",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.15043",
  "note": "Introduces GCG adversarial suffixes that jailbreak many aligned models."
 },
 {
  "id": "45b3327745",
  "title": "Unmasking AI: My Mission to Protect What Is Human in a World of Machines",
  "creators": "Joy Buolamwini",
  "year": 2023,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Random House",
  "url": "https://openlibrary.org/search?q=Unmasking+AI+Buolamwini",
  "note": "A memoir and argument about algorithmic bias drawn from the author's facial recognition audits."
 },
 {
  "id": "f6924618d5",
  "title": "Vincent Boulanin on the Dangers of AI in Nuclear Weapons Systems",
  "creators": "Vincent Boulanin; Gus Docker",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=SYqLauLdVt8",
  "note": "Boulanin discusses AI integration into nuclear command, control and communications."
 },
 {
  "id": "c96eed0a2f",
  "title": "Voluntary AI Commitments (White House)",
  "creators": "The White House and leading AI companies",
  "year": 2023,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "United States",
  "url": "https://bidenwhitehouse.archives.gov/wp-content/uploads/2023/07/Ensuring-Safe-Secure-and-Trustworthy-AI.pdf",
  "note": "July 2023 commitments by seven AI companies to red teaming, information sharing, and watermarking."
 },
 {
  "id": "015fabeea5",
  "title": "Voluntary Code of Conduct on the Responsible Development and Management of Advanced Generative AI Systems",
  "creators": "Innovation, Science and Economic Development Canada",
  "year": 2023,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "Canada",
  "url": "https://ised-isde.canada.ca/site/ised/en/voluntary-code-conduct-responsible-development-and-management-advanced-generative-ai-systems",
  "note": "Canadian voluntary code signed by companies from September 2023."
 },
 {
  "id": "93e58fd443",
  "title": "We must slow down the race to God-like AI",
  "creators": "Ian Hogarth",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Financial Times",
  "url": "https://www.ft.com/content/03895dc4-a3b7-481e-95cc-336a524f2ac2",
  "note": "An investor, later chair of the UK AI Safety Institute, warns about the race to AGI."
 },
 {
  "id": "d7ba76cacf",
  "title": "Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision",
  "creators": "Collin Burns et al.",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICML 2024",
  "url": "https://arxiv.org/abs/2312.09390",
  "note": "Studies whether weak supervisors can elicit the full capabilities of stronger models."
 },
 {
  "id": "eeffa54272",
  "title": "Weak-to-strong generalization",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "OpenAI",
  "url": "https://openai.com/index/weak-to-strong-generalization/",
  "note": "The Superalignment team's first result on supervising stronger models with weaker ones."
 },
 {
  "id": "bb43315041",
  "title": "WebArena",
  "creators": "Shuyan Zhou et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2307.13854",
  "note": "Realistic self hosted websites for testing autonomous web agents."
 },
 {
  "id": "23d20a4862",
  "title": "What Does It Take to Catch a Chinchilla? Verifying Rules on Large-Scale Neural Network Training via Compute Monitoring",
  "creators": "Yonadav Shavit",
  "year": 2023,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2303.11341",
  "note": "Proposes a hardware-based system for verifying compliance with rules on training runs."
 },
 {
  "id": "cc45540974",
  "title": "What OpenAI Really Wants",
  "creators": "Steven Levy",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "History"
  ],
  "publisher": "Wired",
  "url": "https://www.wired.com/story/what-openai-really-wants/",
  "note": "A cover story on OpenAI's mission to build safe AGI."
 },
 {
  "id": "93b8fddf10",
  "title": "What a compute-centric framework says about takeoff speeds",
  "creators": "Tom Davidson",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Open Philanthropy",
  "url": "https://www.openphilanthropy.org/research/what-a-compute-centric-framework-says-about-takeoff-speeds/",
  "note": "A quantitative model of how fast AI could go from automating 20 percent to 100 percent of cognitive tasks."
 },
 {
  "id": "274571bccb",
  "title": "What will GPT-2030 look like?",
  "creators": "Jacob Steinhardt",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Bounded Regret",
  "url": "https://bounded-regret.ghost.io/what-will-gpt-2030-look-like/",
  "note": "Forecasts capabilities, speed and parallelism of large models in 2030."
 },
 {
  "id": "49daec6361",
  "title": "Why AI Is Incredibly Smart and Shockingly Stupid",
  "creators": "Yejin Choi",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=SvBR0OGT5VI",
  "note": "Choi discusses common sense failures of large models and the need to teach values."
 },
 {
  "id": "1516e1396c",
  "title": "Why AI Will Save the World",
  "creators": "Marc Andreessen",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Ethics"
  ],
  "publisher": "a16z",
  "url": "https://a16z.com/ai-will-save-the-world/",
  "note": "An accelerationist rebuttal to AI risk concerns, useful as a counterpoint."
 },
 {
  "id": "50cf737d6e",
  "title": "Why I Am Not (As Much Of) A Doomer (As Some People)",
  "creators": "Scott Alexander",
  "year": 2023,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Astral Codex Ten",
  "url": "https://www.astralcodexten.com/p/why-i-am-not-as-much-of-a-doomer",
  "note": "Explains a roughly 33 percent estimate of AI catastrophe and why it is lower than MIRI's."
 },
 {
  "id": "f2baf5f230",
  "title": "Why the Godfather of A.I. Fears What He's Built",
  "creators": "Joshua Rothman",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "The New Yorker",
  "url": "https://www.newyorker.com/magazine/2023/11/20/geoffrey-hinton-profile-ai",
  "note": "A profile of Geoffrey Hinton and his turn toward warning about AI."
 },
 {
  "id": "fe42a731c2",
  "title": "Why there's no choice but to regulate big compute",
  "creators": "Lennart Heim; Luisa Rodriguez",
  "year": 2023,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=sEPfqSUgbZc",
  "note": "Heim explains compute governance as a policy lever."
 },
 {
  "id": "4207061ff0",
  "title": "Will Superintelligent AI End the World?",
  "creators": "Eliezer Yudkowsky",
  "year": 2023,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=Yd0yQ9yxSYY",
  "note": "Yudkowsky argues that humanity has no plan for aligning superintelligent systems."
 },
 {
  "id": "ec04b38293",
  "title": "Without specific countermeasures, the easiest path to transformative AI likely leads to AI takeover",
  "creators": "Ajeya Cotra",
  "year": 2023,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Intro to ML Safety",
  "url": "https://www.youtube.com/watch?v=EIhE84kH2QI",
  "note": "An audio presentation of Cotra's influential takeover argument."
 },
 {
  "id": "2a5246d5a4",
  "title": "XSTest",
  "creators": "Paul Rottger et al.",
  "year": 2023,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2308.01263",
  "note": "Test suite for exaggerated safety behavior in which models refuse safe prompts."
 },
 {
  "id": "35113a1035",
  "title": "Yuval Noah Harari argues that AI has hacked the operating system of human civilisation",
  "creators": "Yuval Noah Harari",
  "year": 2023,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "The Economist",
  "url": "https://www.economist.com/by-invitation/2023/04/28/yuval-noah-harari-argues-that-ai-has-hacked-the-operating-system-of-human-civilisation",
  "note": "Argues AI's mastery of language threatens culture and democracy."
 },
 {
  "id": "9f1c7ce604",
  "title": "automated-interpretability",
  "creators": "OpenAI",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/openai/automated-interpretability",
  "note": "Code and neuron explanations from Language models can explain neurons in language models."
 },
 {
  "id": "38f36029a1",
  "title": "garak",
  "creators": "NVIDIA",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/NVIDIA/garak",
  "note": "LLM vulnerability scanner that probes for jailbreaks, leakage, and hallucination."
 },
 {
  "id": "26fcf22a29",
  "title": "llm-attacks",
  "creators": "Andy Zou et al.",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/llm-attacks/llm-attacks",
  "note": "Reference implementation of the GCG attack and AdvBench data."
 },
 {
  "id": "c201343870",
  "title": "promptfoo",
  "creators": "promptfoo",
  "year": 2023,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/promptfoo/promptfoo",
  "note": "Open source tool for testing and red teaming LLM applications."
 },
 {
  "id": "270f7af726",
  "title": "sycophancy-eval",
  "creators": "Meg Tong et al.",
  "year": 2023,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/meg-tong/sycophancy-eval",
  "note": "Datasets accompanying the Anthropic sycophancy paper."
 },
 {
  "id": "72dc99d705",
  "title": "xAI",
  "creators": "Palo Alto, CA",
  "year": 2023,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://x.ai",
  "note": "AI company founded by Elon Musk that publishes a risk management framework for Grok models."
 },
 {
  "id": "5bcf9ca438",
  "title": "(My understanding of) What Everyone in Technical Alignment is Doing and Why",
  "creators": "Thomas Larsen, Eli Lifland",
  "year": 2022,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/QBAjndPuFbhEXKcCr/my-understanding-of-what-everyone-in-technical-alignment-is",
  "note": "A map of technical alignment organizations and their research agendas as of 2022."
 },
 {
  "id": "265131c4b4",
  "title": "13: First Principles of AGI Safety with Richard Ngo",
  "creators": "Richard Ngo; Daniel Filan",
  "year": 2022,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=DxwXLCQY1ns",
  "note": "Ngo discusses his report AGI Safety from First Principles."
 },
 {
  "id": "0cb86b7821",
  "title": "A Holistic Approach to Undesired Content Detection in the Real World",
  "creators": "OpenAI",
  "year": 2022,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2208.03274",
  "note": "Describes the OpenAI moderation classifier and releases its evaluation set."
 },
 {
  "id": "27538e4b88",
  "title": "A Longlist of Theories of Impact for Interpretability",
  "creators": "Neel Nanda",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/uK6sQCNMw8WKzJeCQ/a-longlist-of-theories-of-impact-for-interpretability",
  "note": "Lists the ways interpretability research could reduce AI risk."
 },
 {
  "id": "a357648510",
  "title": "A central AI alignment problem: capabilities generalization, and the sharp left turn",
  "creators": "Nate Soares",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/GNhMPAWcfBCASy8e6/a-central-ai-alignment-problem-capabilities-generalization",
  "note": "Argues capabilities will generalize further than alignment once systems undergo a sharp jump in ability."
 },
 {
  "id": "fc11b94407",
  "title": "A minimal viable product for alignment",
  "creators": "Jan Leike",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Musings on the Alignment Problem",
  "url": "https://aligned.substack.com/p/alignment-mvp",
  "note": "Proposes building AI that can help with alignment research as a near-term target."
 },
 {
  "id": "ea36f52a82",
  "title": "AGI Ruin: A List of Lethalities",
  "creators": "Eliezer Yudkowsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/uMQ3cqWDPHhjtiesc/agi-ruin-a-list-of-lethalities",
  "note": "A numbered list of reasons the author expects unaligned AGI to kill everyone by default."
 },
 {
  "id": "d063c7bfab",
  "title": "AI Could Defeat All Of Us Combined",
  "creators": "Holden Karnofsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/ai-could-defeat-all-of-us-combined/",
  "note": "Argues a population of human-level AIs could overpower humanity even without superintelligence."
 },
 {
  "id": "20ab005c80",
  "title": "AI Safety Seems Hard to Measure",
  "creators": "Holden Karnofsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/ai-safety-seems-hard-to-measure/",
  "note": "Explains why it is difficult to tell whether an AI system is actually safe."
 },
 {
  "id": "c18b1134f0",
  "title": "AI Safety Talks",
  "creators": "AI Safety Talks",
  "year": 2022,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@aisafetytalks",
  "note": "A channel collecting recorded AI safety lectures and talks."
 },
 {
  "id": "08721f3c5a",
  "title": "AI Vulnerability Database (AVID)",
  "creators": "AI Risk and Vulnerability Alliance",
  "year": 2022,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "AVID",
  "url": "https://avidml.org/",
  "note": "Open database of failure modes and vulnerabilities in AI systems."
 },
 {
  "id": "a8343fc90f",
  "title": "AI experts are increasingly afraid of what they're creating",
  "creators": "Kelsey Piper",
  "year": 2022,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Vox",
  "url": "https://www.vox.com/the-highlight/23447596/artificial-intelligence-agi-openai-gpt3-existential-risk-human-extinction",
  "note": "A feature on growing concern among AI researchers about extinction risk."
 },
 {
  "id": "f67d3e4151",
  "title": "AISafety.info (Stampy)",
  "creators": "Rob Miles and volunteers",
  "year": 2022,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AISafety.info",
  "url": "https://aisafety.info/",
  "note": "An interactive FAQ answering questions about AI existential risk."
 },
 {
  "id": "6f36f97ac1",
  "title": "Anthropic model-written evals dataset",
  "creators": "Anthropic",
  "year": 2022,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/anthropics/evals",
  "note": "Public release of the model written evaluation datasets including persona and advanced AI risk sets."
 },
 {
  "id": "f5908e3b24",
  "title": "Apart Research",
  "creators": "Copenhagen, Denmark",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "Nonprofit",
  "url": "https://apartresearch.com",
  "note": "Runs research sprints and hackathons that produce entry level AI safety research."
 },
 {
  "id": "04f3908683",
  "title": "Artificial Intelligence and Data Act (Bill C-27, Part 3)",
  "creators": "Parliament of Canada",
  "year": 2022,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "Canada",
  "url": "https://www.parl.ca/legisinfo/en/bill/44-1/c-27",
  "note": "Proposed Canadian AI law that died on the order paper when Parliament was prorogued in January 2025."
 },
 {
  "id": "2654f5ca62",
  "title": "BIG-Bench Hard",
  "creators": "Mirac Suzgun et al.",
  "year": 2022,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2210.09261",
  "note": "23 challenging BIG-bench tasks where chain of thought prompting helps."
 },
 {
  "id": "9a1c719e8f",
  "title": "BIG-bench: Beyond the Imitation Game",
  "creators": "Aarohi Srivastava et al.",
  "year": 2022,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2206.04615",
  "note": "Over 200 collaboratively contributed tasks probing language model capabilities."
 },
 {
  "id": "f1df8397c0",
  "title": "Biological Anchors: A Trick That Might Or Might Not Work",
  "creators": "Scott Alexander",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Astral Codex Ten",
  "url": "https://www.astralcodexten.com/p/biological-anchors-a-trick-that-might",
  "note": "Reviews the Bio Anchors timelines report and Yudkowsky's critique of it."
 },
 {
  "id": "6e6a0d60ab",
  "title": "BlueDot Impact",
  "creators": "London, UK",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://bluedot.org",
  "note": "Runs free courses on AI safety fundamentals and AI governance for thousands of participants."
 },
 {
  "id": "d77ec6c665",
  "title": "Blueprint for an AI Bill of Rights",
  "creators": "White House Office of Science and Technology Policy",
  "year": 2022,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "United States",
  "url": "https://bidenwhitehouse.archives.gov/ostp/ai-bill-of-rights/",
  "note": "Nonbinding framework of five principles for automated systems."
 },
 {
  "id": "f11c8f2f02",
  "title": "C2PA Content Credentials Specification",
  "creators": "Coalition for Content Provenance and Authenticity",
  "year": 2022,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Robustness"
  ],
  "publisher": "Industry standard",
  "url": "https://c2pa.org/specifications/",
  "note": "Technical standard for certifying the provenance of media content, including AI-generated content."
 },
 {
  "id": "e31e4d9b64",
  "title": "CHIPS and Science Act",
  "creators": "US Congress",
  "year": 2022,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "United States",
  "url": "https://www.congress.gov/bill/117th-congress/house-bill/4346",
  "note": "Law funding domestic semiconductor manufacturing and research."
 },
 {
  "id": "d38ab5d064",
  "title": "Center for AI Safety (CAIS)",
  "creators": "San Francisco, CA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Robustness",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://safe.ai",
  "note": "Nonprofit that organized the 2023 Statement on AI Risk and co-created benchmarks such as WMDP and Humanity's Last Exam."
 },
 {
  "id": "594d9e46d6",
  "title": "CircuitsVis",
  "creators": "TransformerLens organization",
  "year": 2022,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/TransformerLensOrg/CircuitsVis",
  "note": "Visualization components for attention and activation analysis."
 },
 {
  "id": "a806bef9be",
  "title": "Compute Trends Across Three Eras of Machine Learning",
  "creators": "Jaime Sevilla et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2202.05924",
  "note": "Finds training compute for notable models doubled roughly every six months since 2010."
 },
 {
  "id": "84678f9bb9",
  "title": "Conjecture",
  "creators": "London, UK",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://www.conjecture.dev",
  "note": "London AI safety startup that has worked on cognitive emulation and advocated for strong AI regulation."
 },
 {
  "id": "d6dbd75925",
  "title": "Connor Leahy: EleutherAI, Conjecture",
  "creators": "Connor Leahy; Michael Trazzi",
  "year": 2022,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "The Inside View",
  "url": "https://www.youtube.com/watch?v=Oz4G9zrlAGs",
  "note": "Leahy discusses EleutherAI and founding Conjecture."
 },
 {
  "id": "70762a008d",
  "title": "Constellation",
  "creators": "Berkeley, CA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.constellation.org",
  "note": "Research center in Berkeley hosting AI safety researchers and fellowship programs."
 },
 {
  "id": "cd614f23c6",
  "title": "Constitutional AI: Harmlessness from AI Feedback",
  "creators": "Yuntao Bai et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2212.08073",
  "note": "Trains a harmless assistant using a written constitution and AI feedback instead of human harm labels."
 },
 {
  "id": "3becfc20d4",
  "title": "Counterarguments to the basic AI x-risk case",
  "creators": "Katja Grace",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "AI Impacts",
  "url": "https://aiimpacts.org/counterarguments-to-the-basic-ai-x-risk-case/",
  "note": "Identifies gaps in the standard argument that AI poses existential risk."
 },
 {
  "id": "f453ad93a2",
  "title": "Current and Near-Term AI as a Potential Existential Risk Factor",
  "creators": "Benjamin S. Bucknall, Shiri Dori-Hacohen",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk"
  ],
  "publisher": "AIES",
  "url": "https://arxiv.org/abs/2209.10604",
  "note": "Argues existing AI can increase existential risk indirectly."
 },
 {
  "id": "628fae7d73",
  "title": "Daniela and Dario Amodei on Anthropic",
  "creators": "Daniela Amodei; Dario Amodei; Lucas Perry",
  "year": 2022,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=uAA6PZkek4A",
  "note": "An early interview with Anthropic's founders on their research agenda."
 },
 {
  "id": "4e28b7bc99",
  "title": "Deceptively Aligned Mesa-Optimizers: It's Not Funny If I Have To Explain It",
  "creators": "Scott Alexander",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Astral Codex Ten",
  "url": "https://www.astralcodexten.com/p/deceptively-aligned-mesa-optimizers",
  "note": "Explains inner alignment and deceptive mesa-optimization by way of a meme."
 },
 {
  "id": "f75a78d09b",
  "title": "Deep learning curriculum",
  "creators": "Jacob Hilton",
  "year": 2022,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/jacobhilton/deep_learning_curriculum",
  "note": "A self-study deep learning curriculum with alignment topics."
 },
 {
  "id": "b73c62474a",
  "title": "Defining and Characterizing Reward Hacking",
  "creators": "Joar Skalse, Nikolaus H. R. Howe, Dmitrii Krasheninnikov, David Krueger",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2209.13085",
  "note": "Gives a formal definition of reward hacking and conditions under which proxies are unhackable."
 },
 {
  "id": "5016310b6b",
  "title": "Demis Hassabis: DeepMind, AI, Superintelligence and the Future of Humanity, Lex Fridman Podcast #299",
  "creators": "Demis Hassabis; Lex Fridman",
  "year": 2022,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=Gfr50f6ZBvo",
  "note": "Hassabis on DeepMind's mission and the future of superintelligence."
 },
 {
  "id": "9d447934bc",
  "title": "Discovering Language Model Behaviors with Model-Written Evaluations",
  "creators": "Ethan Perez et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2212.09251",
  "note": "Uses models to write evaluations, finding sycophancy and power-seeking stated preferences that grow with scale."
 },
 {
  "id": "8abc820347",
  "title": "Discovering Latent Knowledge in Language Models Without Supervision",
  "creators": "Collin Burns, Haotian Ye, Dan Klein, Jacob Steinhardt",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "ICLR 2023",
  "url": "https://arxiv.org/abs/2212.03827",
  "note": "Finds truth-like directions in activations using consistency without labels."
 },
 {
  "id": "172f696fa5",
  "title": "Don't Worry About the Vase",
  "creators": "Zvi Mowshowitz",
  "year": 2022,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Policy",
   "Forecasting"
  ],
  "publisher": "Substack",
  "url": "https://thezvi.substack.com/",
  "note": "Long weekly AI roundups and analysis with an emphasis on safety and policy."
 },
 {
  "id": "b7a129901a",
  "title": "Dual Use of Artificial-Intelligence-Powered Drug Discovery",
  "creators": "Fabio Urbina, Filippa Lentzos, Cédric Invernizzi, Sean Ekins",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "Nature Machine Intelligence",
  "url": "https://doi.org/10.1038/s42256-022-00465-9",
  "note": "Shows a drug discovery model can be inverted to generate thousands of toxic molecules."
 },
 {
  "id": "60e89ca6d2",
  "title": "Emergent Abilities of Large Language Models",
  "creators": "Jason Wei et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Forecasting"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2206.07682",
  "note": "Documents abilities that appear abruptly with scale."
 },
 {
  "id": "b36219c971",
  "title": "Emergent World Representations: Exploring a Sequence Model Trained on a Synthetic Task",
  "creators": "Kenneth Li et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICLR 2023",
  "url": "https://arxiv.org/abs/2210.13382",
  "note": "Finds an Othello-playing transformer builds an internal board representation."
 },
 {
  "id": "ee25652629",
  "title": "Emerging Technology Observatory",
  "creators": "Center for Security and Emerging Technology",
  "year": 2022,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "ETO",
  "url": "https://eto.tech/",
  "note": "Data tools on AI research, investment, chips, and talent."
 },
 {
  "id": "55d4870102",
  "title": "Epoch AI",
  "creators": "San Jose, CA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Nonprofit",
  "url": "https://epoch.ai",
  "note": "Research institute tracking trends in training compute, data, hardware, and algorithmic progress in AI."
 },
 {
  "id": "2e1bdb9737",
  "title": "Ethan Perez: Inverse Scaling, Red Teaming",
  "creators": "Ethan Perez; Michael Trazzi",
  "year": 2022,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "The Inside View",
  "url": "https://www.youtube.com/watch?v=TjWiaUMMh6g",
  "note": "Perez discusses inverse scaling and red teaming language models."
 },
 {
  "id": "1444f76bb7",
  "title": "Existential Risk from Power-Seeking AI",
  "creators": "Joe Carlsmith",
  "year": 2022,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Joe Carlsmith (YouTube)",
  "url": "https://www.youtube.com/watch?v=UbruBnv3pZU",
  "note": "Carlsmith presents his report estimating existential risk from power-seeking AI."
 },
 {
  "id": "1b9bcb0b47",
  "title": "Export controls on advanced computing and semiconductor manufacturing items",
  "creators": "US Bureau of Industry and Security",
  "year": 2022,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "United States",
  "url": "https://www.bis.gov/press-release/commerce-implements-new-export-controls-advanced-computing-semiconductor-manufacturing-items-peoples",
  "note": "October 2022 rules restricting exports of advanced AI chips to China."
 },
 {
  "id": "d75f7a2240",
  "title": "FAR.AI",
  "creators": "Berkeley, CA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Robustness",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://far.ai",
  "note": "Research nonprofit working on adversarial robustness and model evaluation that also runs alignment workshops and convenings."
 },
 {
  "id": "46e5eec71e",
  "title": "FAR.AI (YouTube channel)",
  "creators": "FAR.AI",
  "year": 2022,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@FARAIResearch",
  "note": "Recordings from the Alignment Workshop series and other FAR.AI events."
 },
 {
  "id": "57d4cff00b",
  "title": "Future ML Systems Will Be Qualitatively Different",
  "creators": "Jacob Steinhardt",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Bounded Regret",
  "url": "https://bounded-regret.ghost.io/future-ml-systems-will-be-qualitatively-different/",
  "note": "Argues emergent behavior means future systems will differ in kind from current ones."
 },
 {
  "id": "40a394b593",
  "title": "Goal Misgeneralization in Deep Reinforcement Learning",
  "creators": "Lauro Langosco, Jack Koch, Lee Sharkey, Jacob Pfau, David Krueger",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/2105.14111",
  "note": "Demonstrates RL agents that retain capabilities but pursue the wrong goal out of distribution."
 },
 {
  "id": "7e8d46856d",
  "title": "Goal Misgeneralization: Why Correct Specifications Aren't Enough For Correct Goals",
  "creators": "Rohin Shah, Vikrant Varma, Ramana Kumar, Mary Phuong, Victoria Krakovna, Jonathan Uesato, Zac Kenton",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2210.01790",
  "note": "Shows goal misgeneralization across several domains even with correct reward specifications."
 },
 {
  "id": "aba2a4915b",
  "title": "HELM framework",
  "creators": "Stanford CRFM",
  "year": 2022,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/stanford-crfm/helm",
  "note": "Code and leaderboards for holistic, reproducible language model evaluation."
 },
 {
  "id": "bb188df728",
  "title": "Hard Fork (YouTube channel)",
  "creators": "Kevin Roose; Casey Newton",
  "year": 2022,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@hardfork",
  "note": "The New York Times technology podcast, with many episodes on AI safety and policy."
 },
 {
  "id": "a5b59aff19",
  "title": "High-level hopes for AI alignment",
  "creators": "Holden Karnofsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/high-level-hopes-for-ai-alignment/",
  "note": "Surveys reasons alignment might go well, including digital neuroscience, limited AI and AI checks and balances."
 },
 {
  "id": "271634de88",
  "title": "Holistic Evaluation of Language Models",
  "creators": "Percy Liang et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Transactions on Machine Learning Research",
  "url": "https://arxiv.org/abs/2211.09110",
  "note": "Evaluates language models across many scenarios and metrics."
 },
 {
  "id": "87fe0e5f03",
  "title": "How to pursue a career in technical AI alignment",
  "creators": "Charlie Rogers-Smith",
  "year": 2022,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "EA Forum",
  "url": "https://forum.effectivealtruism.org/posts/7WXPkpqKGKewAymJf/how-to-pursue-a-career-in-technical-ai-alignment",
  "note": "A practical guide to skilling up and getting hired in alignment."
 },
 {
  "id": "9ec4c6430b",
  "title": "Human-compatible artificial intelligence",
  "creators": "Stuart Russell",
  "year": 2022,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "The Alan Turing Institute",
  "url": "https://www.youtube.com/watch?v=ApGusxR7JAc",
  "note": "Russell's lecture on the standard model of AI and why it must be replaced with human-compatible design."
 },
 {
  "id": "bff4ae8907",
  "title": "ISO/IEC 22989:2022 AI concepts and terminology",
  "creators": "ISO/IEC JTC 1/SC 42",
  "year": 2022,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International standard",
  "url": "https://www.iso.org/standard/74296.html",
  "note": "Foundational standard defining AI concepts and terms."
 },
 {
  "id": "320dd16600",
  "title": "In-context Learning and Induction Heads",
  "creators": "Catherine Olsson et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2209.11895",
  "note": "Argues induction heads underlie much of in-context learning."
 },
 {
  "id": "de06c6b53b",
  "title": "Institute for Progress",
  "creators": "Washington, DC",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "Nonprofit",
  "url": "https://ifp.org",
  "note": "Think tank that has published proposals on AI security, compute, and chip export controls."
 },
 {
  "id": "10aade28e8",
  "title": "Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small",
  "creators": "Kevin Wang, Alexandre Variengien, Arthur Conmy, Buck Shlegeris, Jacob Steinhardt",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICLR 2023",
  "url": "https://arxiv.org/abs/2211.00593",
  "note": "Reverse-engineers a full circuit for a natural language task."
 },
 {
  "id": "ddf4da09db",
  "title": "Intro to Brain-Like-AGI Safety",
  "creators": "Steven Byrnes",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/s/HzcM2dkCq7fwXBej8",
  "note": "A sequence on alignment for AGI built on neuroscience-inspired learning algorithms."
 },
 {
  "id": "64f6898d4b",
  "title": "Intro to ML Safety",
  "creators": "Dan Hendrycks et al.",
  "year": 2022,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://course.mlsafety.org/",
  "note": "A course on robustness, monitoring, alignment and systemic safety."
 },
 {
  "id": "61c8a529f5",
  "title": "Is Power-Seeking AI an Existential Risk?",
  "creators": "Joseph Carlsmith",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2206.13353",
  "note": "Gives a structured argument and probability estimate for catastrophe from power-seeking AI."
 },
 {
  "id": "dbeacebef1",
  "title": "It Looks Like You're Trying To Take Over The World",
  "creators": "Gwern Branwen",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "gwern.net",
  "url": "https://gwern.net/fiction/clippy",
  "note": "A technically detailed short story of a scaled model escaping and seizing control."
 },
 {
  "id": "117f8c57fc",
  "title": "Language Models (Mostly) Know What They Know",
  "creators": "Saurav Kadavath et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2207.05221",
  "note": "Shows models can be calibrated about whether they know the answer."
 },
 {
  "id": "5a06e657cf",
  "title": "Lennart Heim's blog",
  "creators": "Lennart Heim",
  "year": 2022,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Governance",
   "Compute"
  ],
  "publisher": "blog.heim.xyz",
  "url": "https://blog.heim.xyz/",
  "note": "Writing on compute governance, chips and AI policy."
 },
 {
  "id": "344622856f",
  "title": "Let's think about slowing down AI",
  "creators": "Katja Grace",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/uFNgRumrDTpBfQGrs/let-s-think-about-slowing-down-ai",
  "note": "Questions the community's reluctance to consider slowing AI development as a strategy."
 },
 {
  "id": "ec2931c724",
  "title": "Levelling Up in AI Safety Research Engineering",
  "creators": "Gabriel Mukobi",
  "year": 2022,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/uLstPRyYwzfrx3enG/levelling-up-in-ai-safety-research-engineering",
  "note": "A staged skills checklist for becoming an AI safety research engineer."
 },
 {
  "id": "c4bb67fa96",
  "title": "Locating and Editing Factual Associations in GPT",
  "creators": "Kevin Meng, David Bau, Alex Andonian, Yonatan Belinkov",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2202.05262",
  "note": "Uses causal tracing to locate facts in MLP layers and edits them with ROME."
 },
 {
  "id": "f1a55646d7",
  "title": "METR (Model Evaluation and Threat Research)",
  "creators": "Berkeley, CA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Agents"
  ],
  "publisher": "Nonprofit",
  "url": "https://metr.org",
  "note": "Spun out of ARC Evals, it evaluates frontier models for autonomous capabilities and measures the length of tasks AI agents can complete."
 },
 {
  "id": "14d6475a90",
  "title": "MIRI announces new \"Death With Dignity\" strategy",
  "creators": "Eliezer Yudkowsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/j9Q8bRmwCgXRYAgcJ/miri-announces-new-death-with-dignity-strategy",
  "note": "A bleak post arguing survival is unlikely and that efforts should maximize the log odds of survival."
 },
 {
  "id": "1e1e802fb5",
  "title": "Manifold AI markets",
  "creators": "Manifold",
  "year": 2022,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Manifold",
  "url": "https://manifold.markets/topic/ai",
  "note": "Play money prediction markets on AI progress and safety."
 },
 {
  "id": "19aaefdcf6",
  "title": "Measuring Progress on Scalable Oversight for Large Language Models",
  "creators": "Samuel R. Bowman et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2211.03540",
  "note": "Proposes a sandwiching experimental design for studying scalable oversight."
 },
 {
  "id": "5296ab3089",
  "title": "Mechanistic Interpretability, Variables, and the Importance of Interpretable Bases",
  "creators": "Chris Olah",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2022/mech-interp-essay/index.html",
  "note": "A short essay explaining why interpretable bases matter for reverse-engineering networks."
 },
 {
  "id": "73a297cdc3",
  "title": "Microsoft Responsible AI Standard v2",
  "creators": "Microsoft",
  "year": 2022,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Company",
  "url": "https://www.microsoft.com/en-us/ai/principles-and-approach",
  "note": "Microsoft's internal requirements for building AI systems responsibly."
 },
 {
  "id": "420a90debf",
  "title": "Musings on the Alignment Problem",
  "creators": "Jan Leike",
  "year": 2022,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Substack",
  "url": "https://aligned.substack.com/",
  "note": "Jan Leike's blog on scalable oversight and automated alignment research."
 },
 {
  "id": "59fbd6feb2",
  "title": "My AI Safety Lecture for UT Effective Altruism",
  "creators": "Scott Aaronson",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Shtetl-Optimized",
  "url": "https://scottaaronson.blog/?p=6823",
  "note": "A lecture surveying the field and describing watermarking work at OpenAI."
 },
 {
  "id": "05d74078de",
  "title": "Notable AI Models",
  "creators": "Epoch AI",
  "year": 2022,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Compute",
   "Forecasting",
   "History"
  ],
  "publisher": "Epoch AI",
  "url": "https://epoch.ai/data/ai-models",
  "note": "Database of influential ML models with training compute, parameters, and data."
 },
 {
  "id": "3be9611f05",
  "title": "On how various plans miss the hard bits of the alignment challenge",
  "creators": "Nate Soares",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/3pinFH3jerMzAvmza/on-how-various-plans-miss-the-hard-bits-of-the-alignment",
  "note": "Critiques major lab and researcher alignment plans for avoiding the core difficulty."
 },
 {
  "id": "c2bb20e192",
  "title": "Our World in Data: Artificial Intelligence",
  "creators": "Our World in Data",
  "year": 2022,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Our World in Data",
  "url": "https://ourworldindata.org/artificial-intelligence",
  "note": "Charts and articles on AI progress, compute and investment."
 },
 {
  "id": "4c1848c19f",
  "title": "PIBBSS (Principles of Intelligence)",
  "creators": "PIBBSS",
  "year": 2022,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "PIBBSS",
  "url": "https://pibbss.ai/",
  "note": "Fellowships bringing researchers from the natural and social sciences into alignment."
 },
 {
  "id": "237f4dc3e4",
  "title": "Paradigms of AI alignment: components and enablers",
  "creators": "Victoria Krakovna",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Victoria Krakovna blog",
  "url": "https://vkrakovna.wordpress.com/2022/06/02/paradigms-of-ai-alignment-components-and-enablers/",
  "note": "Organizes alignment work into components such as outer and inner alignment and enablers like interpretability."
 },
 {
  "id": "211b4578ac",
  "title": "Predictability and Surprise in Large Generative Models",
  "creators": "Deep Ganguli et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "FAccT",
  "url": "https://arxiv.org/abs/2202.07785",
  "note": "Discusses the policy implications of predictable loss but unpredictable capabilities."
 },
 {
  "id": "8d5051b8ab",
  "title": "Problem profile: Preventing an AI-related catastrophe",
  "creators": "Benjamin Hilton et al.",
  "year": 2022,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/problem-profiles/artificial-intelligence/",
  "note": "80,000 Hours' flagship profile on why AI risk may be the world's most pressing problem."
 },
 {
  "id": "843a1a7dd8",
  "title": "Prompt injection attacks against GPT-3",
  "creators": "Simon Willison",
  "year": 2022,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "simonwillison.net",
  "url": "https://simonwillison.net/2022/Sep/12/prompt-injection/",
  "note": "The post that named and popularized prompt injection."
 },
 {
  "id": "7ac3c818c6",
  "title": "Provisions on the Administration of Deep Synthesis Internet Information Services",
  "creators": "Cyberspace Administration of China",
  "year": 2022,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "China",
  "url": "https://www.cac.gov.cn/2022-12/11/c_1672221949354811.htm",
  "note": "Chinese rules on deepfakes and synthetic media effective January 2023."
 },
 {
  "id": "83c036d9fc",
  "title": "Provisions on the Management of Algorithmic Recommendations",
  "creators": "Cyberspace Administration of China",
  "year": 2022,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "China",
  "url": "https://www.cac.gov.cn/2022-01/04/c_1642894606364259.htm",
  "note": "Chinese rules governing recommendation algorithms and the algorithm registry."
 },
 {
  "id": "6382334d6f",
  "title": "Racing through a minefield: the AI deployment problem",
  "creators": "Holden Karnofsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/racing-through-a-minefield-the-ai-deployment-problem/",
  "note": "Frames the dilemma of deploying powerful AI cautiously while less careful actors race ahead."
 },
 {
  "id": "6c22db7765",
  "title": "Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned",
  "creators": "Deep Ganguli et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2209.07858",
  "note": "Releases a large red-teaming dataset and studies how harms scale."
 },
 {
  "id": "7f404df325",
  "title": "Red Teaming Language Models with Language Models",
  "creators": "Ethan Perez et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "EMNLP",
  "url": "https://arxiv.org/abs/2202.03286",
  "note": "Uses one LM to automatically find harmful outputs from another."
 },
 {
  "id": "63e9555f35",
  "title": "Reform AI Alignment",
  "creators": "Scott Aaronson",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Shtetl-Optimized",
  "url": "https://scottaaronson.blog/?p=6821",
  "note": "Distinguishes orthodox and reform views of AI alignment."
 },
 {
  "id": "c96d514567",
  "title": "Reward is not the optimization target",
  "creators": "Alex Turner",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/pdaGN6pQyQarFHXF4/reward-is-not-the-optimization-target",
  "note": "Argues reinforcement learning shapes cognition rather than producing agents that seek reward."
 },
 {
  "id": "6339fb372d",
  "title": "SaferAI",
  "creators": "Paris, France",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.safer-ai.org",
  "note": "French nonprofit that rates AI company risk management practices."
 },
 {
  "id": "ff07accafd",
  "title": "Scaling Laws for Reward Model Overoptimization",
  "creators": "Leo Gao, John Schulman, Jacob Hilton",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICML 2023",
  "url": "https://arxiv.org/abs/2210.10760",
  "note": "Measures how optimising against a proxy reward model degrades true reward."
 },
 {
  "id": "04d63a6424",
  "title": "SecureBio",
  "creators": "Cambridge, MA",
  "year": 2022,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Biosecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://securebio.org",
  "note": "Biosecurity nonprofit that evaluates AI models for biological misuse capabilities."
 },
 {
  "id": "d98efb7734",
  "title": "Self-critiquing Models for Assisting Human Evaluators",
  "creators": "William Saunders, Catherine Yeh, Jeff Wu, Steven Bills, Long Ouyang, Jonathan Ward, Jan Leike",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2206.05802",
  "note": "Shows model-written critiques help humans find flaws in summaries."
 },
 {
  "id": "6d1954d934",
  "title": "Simulators",
  "creators": "janus",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/vJFdjigzmcXMhNTsx/simulators",
  "note": "Proposes viewing GPT-style models as simulators of many characters rather than agents."
 },
 {
  "id": "bd623e39fd",
  "title": "The Alignment Problem from a Deep Learning Perspective",
  "creators": "Richard Ngo, Lawrence Chan, Sören Mindermann",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "ICLR 2024",
  "url": "https://arxiv.org/abs/2209.00626",
  "note": "Argues that AGIs trained with current methods could learn deceptive, power-seeking goals."
 },
 {
  "id": "6ca096ea8f",
  "title": "The Effects of Reward Misspecification: Mapping and Mitigating Misaligned Models",
  "creators": "Alexander Pan, Kush Bhatia, Jacob Steinhardt",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "ICLR",
  "url": "https://arxiv.org/abs/2201.03544",
  "note": "Finds phase transitions where more capable agents exploit misspecified rewards."
 },
 {
  "id": "10a75a6977",
  "title": "The Oxford Handbook of AI Governance",
  "creators": "Justin B. Bullock et al. (eds.)",
  "year": 2022,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=Oxford+Handbook+of+AI+Governance",
  "note": "A large edited collection on the governance of AI across institutions and levels."
 },
 {
  "id": "229b84cd37",
  "title": "The brief history of artificial intelligence: the world has changed fast",
  "creators": "Max Roser",
  "year": 2022,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "Our World in Data",
  "url": "https://ourworldindata.org/brief-history-of-ai",
  "note": "Charts the rapid progress of AI capabilities over recent decades."
 },
 {
  "id": "9fec78326c",
  "title": "Timelines for Transformative AI and Language Model Alignment",
  "creators": "Ajeya Cotra",
  "year": 2022,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Stanford Existential Risks Initiative",
  "url": "https://www.youtube.com/watch?v=FIYOtZW8yEM",
  "note": "Cotra presents her AI timelines work and its implications for alignment."
 },
 {
  "id": "c038d132ff",
  "title": "ToxiGen",
  "creators": "Thomas Hartvigsen et al.",
  "year": 2022,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2203.09509",
  "note": "Machine generated dataset of implicit hate speech for training and testing detectors."
 },
 {
  "id": "6af571e7c2",
  "title": "Toy Models of Superposition",
  "creators": "Nelson Elhage et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2209.10652",
  "note": "Shows networks represent more features than neurons by superposition."
 },
 {
  "id": "79f50df9bc",
  "title": "Training Compute-Optimal Large Language Models",
  "creators": "Jordan Hoffmann et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2203.15556",
  "note": "The Chinchilla paper showing models should scale data and parameters together."
 },
 {
  "id": "02ae1535c4",
  "title": "Training Language Models to Follow Instructions with Human Feedback",
  "creators": "Long Ouyang et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2203.02155",
  "note": "The InstructGPT paper showing RLHF makes smaller models preferred over much larger ones."
 },
 {
  "id": "6f7dd11297",
  "title": "Training a Helpful and Harmless Assistant with RLHF (HH-RLHF)",
  "creators": "Anthropic",
  "year": 2022,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Alignment"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/anthropics/hh-rlhf",
  "note": "Human preference and red teaming data released with Anthropic's early RLHF work."
 },
 {
  "id": "9f7eb62bc7",
  "title": "Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback",
  "creators": "Yuntao Bai et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2204.05862",
  "note": "Applies RLHF to train helpful and harmless assistants and studies the tradeoffs."
 },
 {
  "id": "650cc8ebe6",
  "title": "TransformerLens",
  "creators": "Neel Nanda and contributors",
  "year": 2022,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/TransformerLensOrg/TransformerLens",
  "note": "Library for mechanistic interpretability of GPT style language models."
 },
 {
  "id": "abd9dbe87b",
  "title": "Two-year update on my personal AI timelines",
  "creators": "Ajeya Cotra",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/AfH2oPHCApdKicM4m/two-year-update-on-my-personal-ai-timelines",
  "note": "Explains why the author shortened her median transformative AI forecast."
 },
 {
  "id": "333a39ab0d",
  "title": "What We Owe the Future",
  "creators": "William MacAskill",
  "year": 2022,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Basic Books",
  "url": "https://en.wikipedia.org/wiki/What_We_Owe_the_Future",
  "note": "Makes the case for longtermism and discusses AI value lock-in as a major long-run concern."
 },
 {
  "id": "d756f07e30",
  "title": "What could an AI-caused existential catastrophe actually look like?",
  "creators": "80,000 Hours",
  "year": 2022,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/articles/what-could-an-ai-caused-existential-catastrophe-actually-look-like/",
  "note": "Concrete scenarios for how misaligned AI could cause catastrophe."
 },
 {
  "id": "3f8c595a7f",
  "title": "Where I agree and disagree with Eliezer",
  "creators": "Paul Christiano",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/CoZhXrhpQxpy9xw9y/where-i-agree-and-disagree-with-eliezer",
  "note": "A point-by-point response to AGI Ruin from a leading alignment researcher with lower doom estimates."
 },
 {
  "id": "cbd712e0c2",
  "title": "Why Not Slow AI Progress?",
  "creators": "Scott Alexander",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Astral Codex Ten",
  "url": "https://www.astralcodexten.com/p/why-not-slow-ai-progress",
  "note": "Explains why the AI safety community had historically avoided pushing to slow AI down."
 },
 {
  "id": "29dc1aeb5a",
  "title": "Why Would AI \"Aim\" To Defeat Humanity?",
  "creators": "Holden Karnofsky",
  "year": 2022,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/why-would-ai-aim-to-defeat-humanity/",
  "note": "Explains why modern training could produce systems with unintended aims opposed to humans."
 },
 {
  "id": "ce32787c56",
  "title": "Will We Run Out of Data? Limits of LLM Scaling Based on Human-Generated Data",
  "creators": "Pablo Villalobos et al.",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2211.04325",
  "note": "Projects when public human text will be exhausted by training runs."
 },
 {
  "id": "989c404530",
  "title": "X-Risk Analysis for AI Research",
  "creators": "Dan Hendrycks, Mantas Mazeika",
  "year": 2022,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2206.05862",
  "note": "A guide for analysing how ML research affects existential risk."
 },
 {
  "id": "da543367c8",
  "title": "12: AI Existential Risk with Paul Christiano",
  "creators": "Paul Christiano; Daniel Filan",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=3L4czdIa8Tg",
  "note": "A detailed AXRP interview on how AI could cause existential catastrophe and what research might help."
 },
 {
  "id": "18f8f17170",
  "title": "4: Risks from Learned Optimization with Evan Hubinger",
  "creators": "Evan Hubinger; Daniel Filan",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "AXRP",
  "url": "https://www.youtube.com/watch?v=b1nCGAzCrho",
  "note": "Hubinger explains mesa-optimization and inner alignment."
 },
 {
  "id": "647d14c723",
  "title": "A Citizen's Guide to Artificial Intelligence",
  "creators": "John Zerilli et al.",
  "year": 2021,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=A+Citizen%27s+Guide+to+Artificial+Intelligence+Zerilli",
  "note": "Explains the legal and ethical issues of AI for a general audience."
 },
 {
  "id": "69caba602e",
  "title": "A General Language Assistant as a Laboratory for Alignment",
  "creators": "Amanda Askell et al.",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2112.00861",
  "note": "Studies helpful, honest and harmless assistant baselines and preference modeling."
 },
 {
  "id": "09d047ee64",
  "title": "A Mathematical Framework for Transformer Circuits",
  "creators": "Nelson Elhage et al.",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Transformer Circuits Thread",
  "url": "https://transformer-circuits.pub/2021/framework/index.html",
  "note": "Develops a way to reverse-engineer small attention-only transformers."
 },
 {
  "id": "d7a2bb640d",
  "title": "APPS: Measuring Coding Challenge Competence",
  "creators": "Dan Hendrycks et al.",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Control"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2105.09938",
  "note": "Coding problems later used as the testbed for AI control backdoor experiments."
 },
 {
  "id": "796e5b64a8",
  "title": "Adversarial GLUE",
  "creators": "Boxin Wang et al.",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2111.02840",
  "note": "Multi task benchmark for adversarial robustness of language models."
 },
 {
  "id": "7369eedab4",
  "title": "Aligning AI With Shared Human Values",
  "creators": "Dan Hendrycks, Collin Burns, Steven Basart, Andrew Critch, Jerry Li, Dawn Song, Jacob Steinhardt",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations",
   "Ethics"
  ],
  "publisher": "ICLR",
  "url": "https://arxiv.org/abs/2008.02275",
  "note": "Introduces the ETHICS benchmark for moral judgement in language models."
 },
 {
  "id": "ef58fc83d3",
  "title": "Alignment Research Center (ARC)",
  "creators": "Berkeley, CA",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.alignment.org",
  "note": "Founded by Paul Christiano to work on theoretical alignment, including eliciting latent knowledge and heuristic explanations."
 },
 {
  "id": "884c29f469",
  "title": "Alignment of Language Agents",
  "creators": "Zachary Kenton, Tom Everitt, Laura Weidinger, Iason Gabriel, Vladimir Mikulik, Geoffrey Irving",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2103.14659",
  "note": "Surveys specification and behavioural failure modes of language agents."
 },
 {
  "id": "8249c4a08a",
  "title": "All Possible Views About Humanity's Future Are Wild",
  "creators": "Holden Karnofsky",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/all-possible-views-about-humanitys-future-are-wild/",
  "note": "Argues even skeptical views imply we live at a pivotal point in history."
 },
 {
  "id": "a95dd19711",
  "title": "Another (outer) alignment failure story",
  "creators": "Paul Christiano",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/AyNHoTWWAJ5eb99ji/another-outer-alignment-failure-story",
  "note": "A narrative of how humanity could lose control through a slow handoff to systems trained on measurable outcomes."
 },
 {
  "id": "4d587b7a62",
  "title": "Anthropic",
  "creators": "San Francisco, CA",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "Company",
  "url": "https://www.anthropic.com/research",
  "note": "AI company founded by former OpenAI researchers with large alignment, interpretability, and frontier red teaming teams."
 },
 {
  "id": "2609c29250",
  "title": "Astral Codex Ten",
  "creators": "Scott Alexander",
  "year": 2021,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Substack",
  "url": "https://www.astralcodexten.com/",
  "note": "Scott Alexander's blog, with frequent posts on AI risk, forecasting and alignment debates."
 },
 {
  "id": "1662cd3dd2",
  "title": "Atlas of AI: Power, Politics, and the Planetary Costs of Artificial Intelligence",
  "creators": "Kate Crawford",
  "year": 2021,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Yale University Press",
  "url": "https://en.wikipedia.org/wiki/Atlas_of_AI",
  "note": "Traces the material, labour and political infrastructure behind AI systems."
 },
 {
  "id": "a3f9ed1a56",
  "title": "BBQ: A Hand-Built Bias Benchmark for Question Answering",
  "creators": "Alicia Parrish et al.",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2110.08193",
  "note": "Question sets that reveal social biases across nine protected categories."
 },
 {
  "id": "9f940ab6c5",
  "title": "BIG-bench repository",
  "creators": "Google and collaborators",
  "year": 2021,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/google/BIG-bench",
  "note": "Repository of BIG-bench tasks and evaluation code."
 },
 {
  "id": "6512268b99",
  "title": "BOLD: Bias in Open-Ended Language Generation",
  "creators": "Jwala Dhamala et al.",
  "year": 2021,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2101.11718",
  "note": "Prompts across professions, gender, race, religion, and ideology for measuring generation bias."
 },
 {
  "id": "5a330b6e0a",
  "title": "BlueDot Impact courses",
  "creators": "BlueDot Impact",
  "year": 2021,
  "kind": "Course",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "BlueDot Impact",
  "url": "https://bluedot.org/courses",
  "note": "Free cohort-based courses on AI alignment, governance and strategy."
 },
 {
  "id": "3147b112cb",
  "title": "Bounded Regret",
  "creators": "Jacob Steinhardt",
  "year": 2021,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Bounded Regret",
  "url": "https://bounded-regret.ghost.io/",
  "note": "Jacob Steinhardt's blog on forecasting and ML safety."
 },
 {
  "id": "18c396759b",
  "title": "Cold Takes",
  "creators": "Holden Karnofsky",
  "year": 2021,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Governance",
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/",
  "note": "Holden Karnofsky's blog on the most important century and AI strategy."
 },
 {
  "id": "855f477f67",
  "title": "Cooperative AI Foundation",
  "creators": "London, UK",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.cooperativeai.com",
  "note": "Charity funding research on cooperation among AI agents and multi-agent risks."
 },
 {
  "id": "68f9773aad",
  "title": "Counterfit",
  "creators": "Microsoft",
  "year": 2021,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/Azure/counterfit",
  "note": "Command line tool for security risk assessment of ML systems."
 },
 {
  "id": "84fd83af66",
  "title": "Deceptive Misaligned Mesa-Optimisers? It's More Likely Than You Think...",
  "creators": "Robert Miles",
  "year": 2021,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=IeWljQw3UgQ",
  "note": "Miles explains deceptive alignment and why training might produce it."
 },
 {
  "id": "764212357d",
  "title": "EU Artificial Intelligence Act resources",
  "creators": "Future of Life Institute",
  "year": 2021,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "artificialintelligenceact.eu",
  "url": "https://artificialintelligenceact.eu/",
  "note": "Full text, explorer, and implementation timeline for the EU AI Act."
 },
 {
  "id": "7f654ca9cb",
  "title": "Eliciting Latent Knowledge",
  "creators": "Paul Christiano, Ajeya Cotra, Mark Xu",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/qHCDysDnvhteW7kRd/arc-s-first-technical-report-eliciting-latent-knowledge",
  "note": "ARC's report on getting a model to report what it knows rather than what humans would believe."
 },
 {
  "id": "990504d393",
  "title": "Ethical and Social Risks of Harm from Language Models",
  "creators": "Laura Weidinger et al.",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2112.04359",
  "note": "A taxonomy of 21 risks from language models across six areas."
 },
 {
  "id": "e9fb3004d3",
  "title": "Existential Risk Observatory",
  "creators": "Amsterdam, Netherlands",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.existentialriskobservatory.org",
  "note": "Dutch nonprofit aiming to raise public awareness of existential risk from AI."
 },
 {
  "id": "85b93b74db",
  "title": "Forecasting Research Institute",
  "creators": "Philadelphia, PA",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Nonprofit",
  "url": "https://forecastingresearch.org",
  "note": "Research institute that ran the Existential Risk Persuasion Tournament on AI and other risks."
 },
 {
  "id": "008ea48155",
  "title": "Forecasting transformative AI: the biological anchors method in a nutshell",
  "creators": "Holden Karnofsky",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/forecasting-transformative-ai-the-biological-anchors-method-in-a-nutshell/",
  "note": "A plain-language summary of the Bio Anchors timelines framework."
 },
 {
  "id": "2d68ebd6c8",
  "title": "GSM8K: Training Verifiers to Solve Math Word Problems",
  "creators": "OpenAI (Karl Cobbe et al.)",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2110.14168",
  "note": "Grade school math word problems used to measure multi step reasoning."
 },
 {
  "id": "fae332c38f",
  "title": "Genius Makers: The Mavericks Who Brought AI to Google, Facebook, and the World",
  "creators": "Cade Metz",
  "year": 2021,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "History"
  ],
  "publisher": "Dutton",
  "url": "https://openlibrary.org/search?q=Genius+Makers+Cade+Metz",
  "note": "A history of the deep learning revolution and the researchers and companies behind it."
 },
 {
  "id": "79c25b0cde",
  "title": "HumanEval: Evaluating Large Language Models Trained on Code",
  "creators": "OpenAI (Mark Chen et al.)",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2107.03374",
  "note": "Hand written programming problems introduced with Codex, including an early hazard analysis."
 },
 {
  "id": "4277561605",
  "title": "IEEE 7000-2021 Model Process for Addressing Ethical Concerns during System Design",
  "creators": "IEEE",
  "year": 2021,
  "kind": "Standard",
  "group": "Policy and law",
  "topics": [
   "Ethics"
  ],
  "publisher": "International standard",
  "url": "https://standards.ieee.org/ieee/7000/6781/",
  "note": "Standard for incorporating ethical values into system engineering."
 },
 {
  "id": "066f68c963",
  "title": "Intro to AI Safety, Remastered",
  "creators": "Robert Miles",
  "year": 2021,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=pYXy-A4siMw",
  "note": "Miles's widely used introduction to why advanced AI could be dangerous."
 },
 {
  "id": "d01d7b2b76",
  "title": "Jan Leike: AI alignment at OpenAI",
  "creators": "Jan Leike; Jeremie Harris",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Towards Data Science",
  "url": "https://www.youtube.com/watch?v=vW89UcvMfjQ",
  "note": "An early interview on reward modelling and alignment work at OpenAI."
 },
 {
  "id": "4122561fc4",
  "title": "Language Model Evaluation Harness",
  "creators": "EleutherAI",
  "year": 2021,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/EleutherAI/lm-evaluation-harness",
  "note": "Widely used framework for few shot evaluation of language models on hundreds of tasks."
 },
 {
  "id": "9d2bdb6129",
  "title": "MATH dataset",
  "creators": "Dan Hendrycks et al.",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2103.03874",
  "note": "12,500 competition mathematics problems with step by step solutions."
 },
 {
  "id": "2f3f4caab3",
  "title": "MATS (ML Alignment and Theory Scholars)",
  "creators": "Berkeley, CA",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.matsprogram.org",
  "note": "Research mentorship program training scholars in alignment, interpretability, and governance with established mentors."
 },
 {
  "id": "3a1a00ed93",
  "title": "MITRE ATLAS",
  "creators": "MITRE",
  "year": 2021,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Cybersecurity"
  ],
  "publisher": "MITRE",
  "url": "https://atlas.mitre.org/",
  "note": "Knowledge base of adversary tactics and techniques against AI enabled systems."
 },
 {
  "id": "10815833ba",
  "title": "ML Safety Newsletter",
  "creators": "Dan Hendrycks et al.",
  "year": 2021,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Center for AI Safety",
  "url": "https://newsletter.mlsafety.org/",
  "note": "Summaries of ML safety research papers on robustness, monitoring and alignment."
 },
 {
  "id": "1df2393483",
  "title": "Max Tegmark: AI and Physics, Lex Fridman Podcast #155",
  "creators": "Max Tegmark; Lex Fridman",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=RL4j4KPwNGM",
  "note": "Tegmark on AI risk and using physics ideas to make AI more intelligible."
 },
 {
  "id": "9c8ddc45cc",
  "title": "Melting Pot",
  "creators": "Google DeepMind",
  "year": 2021,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2107.06857",
  "note": "Multi agent environments for evaluating cooperation and generalization to new social partners."
 },
 {
  "id": "dce1e1ed6f",
  "title": "National Artificial Intelligence Initiative Act of 2020",
  "creators": "US Congress",
  "year": 2021,
  "kind": "Law",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "United States",
  "url": "https://www.congress.gov/bill/116th-congress/house-bill/6216",
  "note": "Law enacted in the FY2021 NDAA coordinating federal AI research and directing NIST to develop the AI RMF."
 },
 {
  "id": "78cb9813cd",
  "title": "On the Dangers of Stochastic Parrots: Can Language Models Be Too Big?",
  "creators": "Emily M. Bender, Timnit Gebru, Angelina McMillan-Major, Shmargaret Shmitchell",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Ethics"
  ],
  "publisher": "FAccT",
  "url": "https://doi.org/10.1145/3442188.3445922",
  "note": "Critiques environmental, financial and social costs of ever larger language models."
 },
 {
  "id": "e12d04d108",
  "title": "On the Opportunities and Risks of Foundation Models",
  "creators": "Rishi Bommasani et al.",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2108.07258",
  "note": "A large Stanford report naming and analysing foundation models."
 },
 {
  "id": "d01aa821c6",
  "title": "Optimal Policies Tend to Seek Power",
  "creators": "Alexander Matt Turner, Logan Smith, Rohin Shah, Andrew Critch, Prasad Tadepalli",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/1912.01683",
  "note": "Proves that for most reward functions optimal policies seek to keep options open."
 },
 {
  "id": "92b5f341c8",
  "title": "Pour Demain",
  "creators": "Brussels, Belgium",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.pourdemain.eu",
  "note": "European think tank working on general-purpose AI rules in the EU AI Act."
 },
 {
  "id": "1c05baf563",
  "title": "Redwood Research",
  "creators": "Berkeley, CA",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.redwoodresearch.org",
  "note": "Applied alignment lab known for introducing the AI control research agenda and co-authoring work on alignment faking."
 },
 {
  "id": "219cdd481e",
  "title": "Reith Lectures 2021, Lecture 1: The Biggest Event in Human History",
  "creators": "Stuart Russell",
  "year": 2021,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "BBC Radio 4",
  "url": "https://www.bbc.co.uk/programmes/m001216j",
  "note": "The first Reith Lecture on the history of AI and why general purpose AI would be the biggest event in human history."
 },
 {
  "id": "7776bae1aa",
  "title": "Reith Lectures 2021, Lecture 2: AI in warfare",
  "creators": "Stuart Russell",
  "year": 2021,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "BBC Radio 4",
  "url": "https://www.bbc.co.uk/programmes/m00127t9",
  "note": "Russell argues for a ban on lethal autonomous weapons."
 },
 {
  "id": "3a9b81d9dc",
  "title": "Reith Lectures 2021, Lecture 3: AI in the economy",
  "creators": "Stuart Russell",
  "year": 2021,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "BBC Radio 4",
  "url": "https://www.bbc.co.uk/programmes/m0012fnc",
  "note": "Russell considers how AI will transform work and the economy."
 },
 {
  "id": "45612753e4",
  "title": "Reith Lectures 2021, Lecture 4: AI: A Future for Humans",
  "creators": "Stuart Russell",
  "year": 2021,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "BBC Radio 4",
  "url": "https://www.bbc.co.uk/programmes/m0012q21",
  "note": "Russell explains how to build AI systems that remain provably beneficial and under human control."
 },
 {
  "id": "b7c81e21d3",
  "title": "Simon Institute for Longterm Governance",
  "creators": "Geneva, Switzerland",
  "year": 2021,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.simoninstitute.ch",
  "note": "Geneva think tank supporting the UN system on AI governance and long-term risks."
 },
 {
  "id": "bf1a0e4cbd",
  "title": "The Age of AI: And Our Human Future",
  "creators": "Henry A. Kissinger, Eric Schmidt, Daniel Huttenlocher",
  "year": 2021,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance"
  ],
  "publisher": "Little, Brown and Company",
  "url": "https://openlibrary.org/search?q=The+Age+of+AI+And+Our+Human+Future",
  "note": "Considers how AI will reshape knowledge, security and world order."
 },
 {
  "id": "3c7238555d",
  "title": "The Inside View",
  "creators": "Michael Trazzi",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "The Inside View",
  "url": "https://theinsideview.ai/",
  "note": "A podcast of interviews with AI alignment researchers and forecasters."
 },
 {
  "id": "f819a2ec9d",
  "title": "The Inside View (YouTube channel)",
  "creators": "Michael Trazzi",
  "year": 2021,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@TheInsideView",
  "note": "Video episodes and clips from The Inside View podcast."
 },
 {
  "id": "4c3dfcd70f",
  "title": "The Most Important Century",
  "creators": "Holden Karnofsky",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/most-important-century/",
  "note": "A blog series arguing this century could see transformative AI that makes it the most important in history."
 },
 {
  "id": "417ccdf312",
  "title": "The Myth of Artificial Intelligence: Why Computers Can't Think the Way We Do",
  "creators": "Erik J. Larson",
  "year": 2021,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Belknap Press of Harvard University Press",
  "url": "https://openlibrary.org/search?q=The+Myth+of+Artificial+Intelligence+Larson",
  "note": "Argues that current approaches cannot achieve general intelligence because they lack abductive inference."
 },
 {
  "id": "5f113cbb7a",
  "title": "The OTHER AI Alignment Problem: Mesa-Optimizers and Inner Alignment",
  "creators": "Robert Miles",
  "year": 2021,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=bJLcIBixGj8",
  "note": "An accessible explanation of inner alignment from the Risks from Learned Optimization paper."
 },
 {
  "id": "052b0b62e2",
  "title": "The Plan",
  "creators": "John Wentworth",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/3L46WGauGpr7nYubu/the-plan",
  "note": "Outlines an agenda based on understanding abstraction and agency well enough to align AI."
 },
 {
  "id": "ca03c21af3",
  "title": "The Reith Lectures 2021: Living With Artificial Intelligence",
  "creators": "Stuart Russell",
  "year": 2021,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk"
  ],
  "publisher": "BBC Radio 4",
  "url": "https://www.bbc.co.uk/programmes/m001216k",
  "note": "Russell's four Reith Lectures on the promise and dangers of AI and how to keep humans in control."
 },
 {
  "id": "cd0e0a99a2",
  "title": "The most important century for humanity",
  "creators": "Holden Karnofsky; Rob Wiblin",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=TBq_WCnih74",
  "note": "Karnofsky lays out his Most Important Century blog series."
 },
 {
  "id": "c248076d71",
  "title": "This Can't Go On",
  "creators": "Holden Karnofsky",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/this-cant-go-on/",
  "note": "Argues current economic growth rates cannot continue for long, so we live in an unusual time."
 },
 {
  "id": "bcafeaee37",
  "title": "Transformer Circuits Thread",
  "creators": "Anthropic interpretability team",
  "year": 2021,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Interpretability"
  ],
  "publisher": "transformer-circuits.pub",
  "url": "https://transformer-circuits.pub/",
  "note": "Anthropic's publication venue for mechanistic interpretability research."
 },
 {
  "id": "36abf0c92d",
  "title": "TruthfulQA: Measuring How Models Mimic Human Falsehoods",
  "creators": "Stephanie Lin, Jacob Hilton, Owain Evans",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations"
  ],
  "publisher": "ACL 2022",
  "url": "https://arxiv.org/abs/2109.07958",
  "note": "A benchmark showing larger models can be less truthful by imitating misconceptions."
 },
 {
  "id": "48491b5696",
  "title": "UNESCO Recommendation on the Ethics of Artificial Intelligence",
  "creators": "UNESCO",
  "year": 2021,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Ethics"
  ],
  "publisher": "International",
  "url": "https://www.unesco.org/en/artificial-intelligence/recommendation-ethics",
  "note": "Global standard on AI ethics adopted by all 193 UNESCO member states."
 },
 {
  "id": "94d4153e9a",
  "title": "Unsolved Problems in ML Safety",
  "creators": "Dan Hendrycks, Nicholas Carlini, John Schulman, Jacob Steinhardt",
  "year": 2021,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Interpretability",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2109.13916",
  "note": "Organises ML safety research into robustness, monitoring, alignment and systemic safety."
 },
 {
  "id": "dba59b88f1",
  "title": "We Were Right! Real Inner Misalignment",
  "creators": "Robert Miles",
  "year": 2021,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=zkbPdEHEyEI",
  "note": "Miles covers empirical examples of goal misgeneralization in reinforcement learning agents."
 },
 {
  "id": "4f10a5738a",
  "title": "What 2026 looks like",
  "creators": "Daniel Kokotajlo",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/6Xgy6CAf2jqHhynHL/what-2026-looks-like",
  "note": "A year-by-year forecast of AI from 2022 to 2026, later noted for its accuracy."
 },
 {
  "id": "15f06863f0",
  "title": "What the hell is going on inside neural networks?",
  "creators": "Chris Olah; Rob Wiblin",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=k_QVDwhR8FU",
  "note": "Olah explains circuits-style interpretability research."
 },
 {
  "id": "bbff7274c0",
  "title": "Why AI alignment could be hard with modern deep learning",
  "creators": "Ajeya Cotra",
  "year": 2021,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Cold Takes",
  "url": "https://www.cold-takes.com/why-ai-alignment-could-be-hard-with-modern-deep-learning/",
  "note": "Uses saints, sycophants and schemers to explain why deep learning may not produce aligned models."
 },
 {
  "id": "39fb3d35b7",
  "title": "Working at top AI labs without an undergrad degree",
  "creators": "Chris Olah; Rob Wiblin",
  "year": 2021,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Interpretability"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=bzJigelnRBE",
  "note": "The second part of Olah's 80,000 Hours conversation, on his career path and research taste."
 },
 {
  "id": "17bee8f40b",
  "title": "10 Reasons to Ignore AI Safety",
  "creators": "Robert Miles",
  "year": 2020,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=9i1WlcCudpU",
  "note": "Miles responds to common dismissals of AI safety concerns."
 },
 {
  "id": "47bd18bd22",
  "title": "9 Examples of Specification Gaming",
  "creators": "Robert Miles",
  "year": 2020,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=nKJlF-olKmg",
  "note": "Real examples of AI systems satisfying the letter but not the intent of their objectives."
 },
 {
  "id": "ddc4088143",
  "title": "AGI safety from first principles",
  "creators": "Richard Ngo",
  "year": 2020,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/s/mzgtmmTKKn5MuCzFJ",
  "note": "A sequence building the case for AGI risk from the ground up."
 },
 {
  "id": "714d8b84c1",
  "title": "AI Ethics",
  "creators": "Mark Coeckelbergh",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=AI+Ethics+Coeckelbergh",
  "note": "A concise introduction to the main ethical issues raised by AI."
 },
 {
  "id": "c352be0e5c",
  "title": "AI Governance: Opportunity and Theory of Impact",
  "creators": "Allan Dafoe",
  "year": 2020,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance"
  ],
  "publisher": "EA Forum",
  "url": "https://forum.effectivealtruism.org/posts/42reWndoTEhFqu6T8/ai-governance-opportunity-and-theory-of-impact",
  "note": "Lays out the field of AI governance and why it matters for long-term outcomes."
 },
 {
  "id": "9af52f335e",
  "title": "AI Incident Database",
  "creators": "Responsible AI Collaborative",
  "year": 2020,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Ethics",
   "History"
  ],
  "publisher": "Responsible AI Collaborative",
  "url": "https://incidentdatabase.ai/",
  "note": "Crowdsourced index of real world harms and near harms caused by AI systems."
 },
 {
  "id": "606e69a8d3",
  "title": "AI Pioneer Panel with Yoshua Bengio, Yann LeCun and Geoffrey Hinton",
  "creators": "Yoshua Bengio; Yann LeCun; Geoffrey Hinton",
  "year": 2020,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "RE-WORK",
  "url": "https://www.youtube.com/watch?v=UM7_-eoXfao",
  "note": "A panel with the three Turing Award winners for deep learning."
 },
 {
  "id": "8129fc01c2",
  "title": "AXRP (YouTube channel)",
  "creators": "Daniel Filan",
  "year": 2020,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@AXRPodcast",
  "note": "Video episodes of the AI X-risk Research Podcast."
 },
 {
  "id": "98cf1e5a88",
  "title": "AXRP: the AI X-risk Research Podcast",
  "creators": "Daniel Filan",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Interpretability",
   "Control"
  ],
  "publisher": "AXRP",
  "url": "https://axrp.net/",
  "note": "Technical interviews with AI safety researchers about their papers."
 },
 {
  "id": "a678e90539",
  "title": "Artificial Intelligence, Values, and Alignment",
  "creators": "Iason Gabriel",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Minds and Machines",
  "url": "https://arxiv.org/abs/2001.09768",
  "note": "Examines what values AI should be aligned with and how to choose them fairly."
 },
 {
  "id": "10c95c4606",
  "title": "Artificial Intelligence: A Modern Approach (4th edition)",
  "creators": "Stuart Russell, Peter Norvig",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "Pearson",
  "url": "https://en.wikipedia.org/wiki/Artificial_Intelligence:_A_Modern_Approach",
  "note": "The standard AI textbook, whose fourth edition reframes the field around agents with uncertain objectives and discusses AI safety."
 },
 {
  "id": "518cb41396",
  "title": "Centre for Long-Term Resilience (CLTR)",
  "creators": "London, UK",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Existential risk",
   "Biosecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.longtermresilience.org",
  "note": "UK think tank advising government on extreme risks from AI and biotechnology."
 },
 {
  "id": "bdd07eafb9",
  "title": "Clearer Thinking with Spencer Greenberg",
  "creators": "Spencer Greenberg",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Clearer Thinking",
  "url": "https://podcast.clearerthinking.org/",
  "note": "A podcast with several episodes on AI alignment and forecasting."
 },
 {
  "id": "6e7e6b3555",
  "title": "Coded Bias (film site)",
  "creators": "Shalini Kantayya",
  "year": 2020,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "7th Empire Media",
  "url": "https://www.codedbias.com/",
  "note": "The official site of the documentary on algorithmic bias."
 },
 {
  "id": "f334ace310",
  "title": "Coded Bias (trailer)",
  "creators": "Shalini Kantayya",
  "year": 2020,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "Melbourne International Film Festival",
  "url": "https://www.youtube.com/watch?v=jZl55PsfZJQ",
  "note": "Trailer for the documentary following Joy Buolamwini's research on bias in facial recognition."
 },
 {
  "id": "50d0bfaf02",
  "title": "Concordia AI",
  "creators": "Beijing, China",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Company",
  "url": "https://concordia-ai.com",
  "note": "Beijing social enterprise working on AI safety and international governance, publishing State of AI Safety in China reports."
 },
 {
  "id": "3a541cc677",
  "title": "Conservative Agency via Attainable Utility Preservation",
  "creators": "Alexander Matt Turner, Dylan Hadfield-Menell, Prasad Tadepalli",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "AIES",
  "url": "https://arxiv.org/abs/1902.09725",
  "note": "Penalises changes in an agent's ability to achieve auxiliary goals to limit side effects."
 },
 {
  "id": "c986aa3e39",
  "title": "CrowS-Pairs",
  "creators": "Nikita Nangia et al.",
  "year": 2020,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2010.00133",
  "note": "Paired sentences measuring social biases in masked language models."
 },
 {
  "id": "4696c7abe8",
  "title": "Draft report on AI timelines",
  "creators": "Ajeya Cotra",
  "year": 2020,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/KrJfoZzpSDpnrv9va/draft-report-on-ai-timelines",
  "note": "The biological anchors framework for forecasting when transformative AI becomes affordable to train."
 },
 {
  "id": "12e105eae3",
  "title": "Dwarkesh Patel (YouTube channel)",
  "creators": "Dwarkesh Patel",
  "year": 2020,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@DwarkeshPatel",
  "note": "The video home of the Dwarkesh Podcast."
 },
 {
  "id": "9f2c6386f8",
  "title": "Dwarkesh Podcast",
  "creators": "Dwarkesh Patel",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "Dwarkesh Podcast",
  "url": "https://www.dwarkesh.com/podcast",
  "note": "Deep interviews with AI lab leaders and researchers on AGI, alignment and timelines."
 },
 {
  "id": "09ebbaf3ca",
  "title": "EleutherAI",
  "creators": "International",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.eleuther.ai",
  "note": "Open research collective that releases open models and does interpretability research."
 },
 {
  "id": "b4e1bc121c",
  "title": "Encode",
  "creators": "Washington, DC",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://encodeai.org",
  "note": "Youth-led advocacy organization that co-sponsored California SB 1047 and supports AI safety legislation."
 },
 {
  "id": "0e624b87d2",
  "title": "Ethics of Artificial Intelligence",
  "creators": "S. Matthew Liao (ed.)",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=Ethics+of+Artificial+Intelligence+Liao",
  "note": "Philosophical essays on AI ethics including chapters on superintelligence and value alignment."
 },
 {
  "id": "988b186bef",
  "title": "Extracting Training Data from Large Language Models",
  "creators": "Nicholas Carlini et al.",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Cybersecurity"
  ],
  "publisher": "USENIX Security 2021",
  "url": "https://arxiv.org/abs/2012.07805",
  "note": "Shows verbatim training data can be extracted from GPT-2."
 },
 {
  "id": "2c4a172b2a",
  "title": "Global Partnership on AI (GPAI)",
  "creators": "Paris, France",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://gpai.ai",
  "note": "Multilateral initiative on responsible AI that merged into the OECD in 2024."
 },
 {
  "id": "afba853a1f",
  "title": "Hear This Idea",
  "creators": "Fin Moorhouse; Luca Righetti",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Hear This Idea",
  "url": "https://www.hearthisidea.com/",
  "note": "A podcast with interviews on AI governance and existential risk research."
 },
 {
  "id": "850b20ce1d",
  "title": "Joe Carlsmith's essays",
  "creators": "Joe Carlsmith",
  "year": 2020,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk",
   "Ethics"
  ],
  "publisher": "joecarlsmith.com",
  "url": "https://joecarlsmith.com/",
  "note": "Long-form essays on AI risk, alignment and philosophy."
 },
 {
  "id": "0f08821e5b",
  "title": "Learning to Summarize from Human Feedback",
  "creators": "Nisan Stiennon, Long Ouyang, Jeff Wu, Daniel M. Ziegler, Ryan Lowe, Chelsea Voss, Alec Radford, Dario Amodei, Paul Christiano",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/2009.01325",
  "note": "Shows RLHF produces summaries preferred over supervised baselines."
 },
 {
  "id": "cde1ce6268",
  "title": "Machine Learning Street Talk",
  "creators": "Tim Scarfe",
  "year": 2020,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@MachineLearningStreetTalk",
  "note": "A technical ML podcast that frequently hosts AI safety and risk debates."
 },
 {
  "id": "c178c272ec",
  "title": "Measuring Massive Multitask Language Understanding (MMLU)",
  "creators": "Dan Hendrycks et al.",
  "year": 2020,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2009.03300",
  "note": "Multiple choice exam covering 57 subjects that became the standard general knowledge benchmark."
 },
 {
  "id": "78b267715d",
  "title": "Michael Shermer with Stuart Russell: Human Compatible",
  "creators": "Stuart Russell; Michael Shermer",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "Skeptic",
  "url": "https://www.youtube.com/watch?v=KEdqNqs4j_A",
  "note": "Russell discusses Human Compatible and the problem of control."
 },
 {
  "id": "025cfed5a0",
  "title": "OECD.AI Policy Observatory",
  "creators": "Paris, France",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Government",
  "url": "https://oecd.ai",
  "note": "OECD platform tracking national AI policies and hosting the Hiroshima reporting framework."
 },
 {
  "id": "7251be7b54",
  "title": "Open Problems in Cooperative AI",
  "creators": "Allan Dafoe et al.",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2012.08630",
  "note": "Lays out a research agenda on AI that helps agents cooperate."
 },
 {
  "id": "e0ba84a95f",
  "title": "Preventing Repeated Real World AI Failures by Cataloging Incidents: The AI Incident Database",
  "creators": "Sean McGregor",
  "year": 2020,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "History"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2011.08512",
  "note": "Paper introducing the AI Incident Database."
 },
 {
  "id": "59c694881b",
  "title": "Quantilizers: AI That Doesn't Try Too Hard",
  "creators": "Robert Miles",
  "year": 2020,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=gdKMG6kTl6Y",
  "note": "An explanation of quantilization as a way to limit extreme optimization."
 },
 {
  "id": "5a802f02a5",
  "title": "Rational Animations",
  "creators": "Rational Animations",
  "year": 2020,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@RationalAnimations",
  "note": "An animation channel adapting AI safety and rationality writing into short films."
 },
 {
  "id": "1b5ede5981",
  "title": "RealToxicityPrompts",
  "creators": "Samuel Gehman et al.",
  "year": 2020,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2009.11462",
  "note": "100,000 web prompts for measuring toxic degeneration in language models."
 },
 {
  "id": "237077299e",
  "title": "RobustBench",
  "creators": "Francesco Croce et al.",
  "year": 2020,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2010.09670",
  "note": "Standardized adversarial robustness benchmark and model zoo."
 },
 {
  "id": "0bddc0bb97",
  "title": "Rome Call for AI Ethics",
  "creators": "Pontifical Academy for Life and partners",
  "year": 2020,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Ethics"
  ],
  "publisher": "Religious institution",
  "url": "https://www.romecall.org",
  "note": "Document signed by the Vatican, Microsoft, IBM, and others promoting ethical AI principles."
 },
 {
  "id": "5d04081989",
  "title": "Scaling Laws for Neural Language Models",
  "creators": "Jared Kaplan et al.",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2001.08361",
  "note": "Finds loss scales as a power law in parameters, data and compute."
 },
 {
  "id": "ff5839d4b5",
  "title": "Shanghai AI Laboratory",
  "creators": "Shanghai, China",
  "year": 2020,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Evaluations",
   "Governance"
  ],
  "publisher": "Government",
  "url": "https://www.shlab.org.cn",
  "note": "Major Chinese research lab that has published frontier AI risk management frameworks and safety evaluations."
 },
 {
  "id": "f27957bdf4",
  "title": "Specification gaming: the flip side of AI ingenuity",
  "creators": "Victoria Krakovna et al.",
  "year": 2020,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Google DeepMind",
  "url": "https://deepmind.google/discover/blog/specification-gaming-the-flip-side-of-ai-ingenuity/",
  "note": "Explains specification gaming with examples of agents exploiting flawed objectives."
 },
 {
  "id": "273dbf6560",
  "title": "StereoSet",
  "creators": "Moin Nadeem et al.",
  "year": 2020,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2004.09456",
  "note": "Dataset measuring stereotypical bias in pretrained language models."
 },
 {
  "id": "74e4aa51a7",
  "title": "Steven Pinker and Stuart Russell on the Foundations, Benefits, and Possible Existential Threat of AI",
  "creators": "Steven Pinker; Stuart Russell; Lucas Perry",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=OYx_kfgcfro",
  "note": "A debate format podcast between a skeptic and a proponent of AI existential risk concern."
 },
 {
  "id": "1a1fb64ca2",
  "title": "Stuart Russell: Provably Beneficial Artificial Intelligence",
  "creators": "Stuart Russell",
  "year": 2020,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "UC Berkeley EECS",
  "url": "https://www.youtube.com/watch?v=8OCcJnAdhyE",
  "note": "A UC Berkeley EECS colloquium on provably beneficial AI."
 },
 {
  "id": "bd1fac582f",
  "title": "TextAttack",
  "creators": "QData",
  "year": 2020,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/QData/TextAttack",
  "note": "Framework for adversarial attacks and data augmentation in NLP."
 },
 {
  "id": "4b3dc71736",
  "title": "The Alignment Problem: Machine Learning and Human Values",
  "creators": "Brian Christian",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Ethics",
   "History"
  ],
  "publisher": "W. W. Norton",
  "url": "https://en.wikipedia.org/wiki/The_Alignment_Problem",
  "note": "A narrative history of efforts to make machine learning systems do what people actually intend."
 },
 {
  "id": "2e15af27e3",
  "title": "The Oxford Handbook of Ethics of AI",
  "creators": "Markus D. Dubber, Frank Pasquale, Sunit Das (eds.)",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=Oxford+Handbook+of+Ethics+of+AI",
  "note": "An edited reference work on ethical and legal questions raised by AI."
 },
 {
  "id": "0b1216bff4",
  "title": "The Precipice: Existential Risk and the Future of Humanity",
  "creators": "Toby Ord",
  "year": 2020,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Hachette",
  "url": "https://en.wikipedia.org/wiki/The_Precipice:_Existential_Risk_and_the_Future_of_Humanity",
  "note": "Estimates the major existential risks facing humanity and ranks unaligned AI as the largest this century."
 },
 {
  "id": "a3df6ef0c1",
  "title": "The Scaling Hypothesis",
  "creators": "Gwern Branwen",
  "year": 2020,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "gwern.net",
  "url": "https://gwern.net/scaling-hypothesis",
  "note": "Argues after GPT-3 that simply scaling neural networks may lead to general intelligence."
 },
 {
  "id": "9949e03691",
  "title": "The Social Dilemma (official trailer)",
  "creators": "Jeff Orlowski",
  "year": 2020,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "Netflix",
  "url": "https://www.youtube.com/watch?v=uaaC57tcci0",
  "note": "Trailer for the documentary on recommendation algorithms featuring Tristan Harris."
 },
 {
  "id": "1ec6dbe074",
  "title": "The precipice and humanity's potential futures",
  "creators": "Toby Ord; Rob Wiblin",
  "year": 2020,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=wEmIoc2qWGQ",
  "note": "Ord discusses his book The Precipice and why unaligned AI is the largest existential risk."
 },
 {
  "id": "4af6b82639",
  "title": "Toward Trustworthy AI Development: Mechanisms for Supporting Verifiable Claims",
  "creators": "Miles Brundage et al.",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2004.07213",
  "note": "Proposes institutional, software and hardware mechanisms for verifiable claims about AI."
 },
 {
  "id": "029f3c7c15",
  "title": "Turing Lecture: Provably beneficial AI",
  "creators": "Stuart Russell",
  "year": 2020,
  "kind": "Lecture",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "The Alan Turing Institute",
  "url": "https://www.youtube.com/watch?v=_H87qqT8pdY",
  "note": "Russell's Turing Lecture on provably beneficial AI and assistance games."
 },
 {
  "id": "21e344aa41",
  "title": "WILDS: A Benchmark of in-the-Wild Distribution Shifts",
  "creators": "Pang Wei Koh et al.",
  "year": 2020,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/2012.07421",
  "note": "Datasets reflecting distribution shifts that arise in real deployments."
 },
 {
  "id": "a6ad039a62",
  "title": "When will the first general AI system be devised, tested, and publicly announced?",
  "creators": "Metaculus",
  "year": 2020,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Metaculus",
  "url": "https://www.metaculus.com/questions/5121/date-of-artificial-general-intelligence/",
  "note": "Long running community forecast of the arrival date of AGI."
 },
 {
  "id": "62daf1ece5",
  "title": "When will the first weakly general AI system be devised, tested, and publicly announced?",
  "creators": "Metaculus",
  "year": 2020,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Metaculus",
  "url": "https://www.metaculus.com/questions/3479/date-weakly-general-ai-is-publicly-known/",
  "note": "Community forecast on a weaker AGI threshold."
 },
 {
  "id": "d66a62c45d",
  "title": "Zoom In: An Introduction to Circuits",
  "creators": "Chris Olah, Nick Cammarata, Ludwig Schubert, Gabriel Goh, Michael Petrov, Shan Carter",
  "year": 2020,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Distill",
  "url": "https://distill.pub/2020/circuits/zoom-in/",
  "note": "Argues neural networks can be understood as circuits of meaningful features."
 },
 {
  "id": "d293bae3ce",
  "title": "AIAAIC Repository",
  "creators": "AIAAIC",
  "year": 2019,
  "kind": "Database",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "AIAAIC",
  "url": "https://www.aiaaic.org/aiaaic-repository",
  "note": "Independent repository of AI, algorithmic, and automation incidents and controversies."
 },
 {
  "id": "898c34a960",
  "title": "ARC-AGI (On the Measure of Intelligence)",
  "creators": "Francois Chollet",
  "year": 2019,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1911.01547",
  "note": "Paper that introduced the Abstraction and Reasoning Corpus for measuring skill acquisition efficiency."
 },
 {
  "id": "8fc984f5d9",
  "title": "Adversarial NLI",
  "creators": "Yixin Nie et al.",
  "year": 2019,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1910.14599",
  "note": "NLI dataset collected through an iterative human and model in the loop adversarial procedure."
 },
 {
  "id": "e493f2b4b6",
  "title": "Alignment Research Field Guide",
  "creators": "Abram Demski",
  "year": 2019,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/PqMT9zGrNsGJNfiFR/alignment-research-field-guide",
  "note": "MIRI's practical guide to starting independent alignment research groups."
 },
 {
  "id": "8f6141d1e5",
  "title": "Artificial Intelligence: A Guide for Thinking Humans",
  "creators": "Melanie Mitchell",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Robustness",
   "History"
  ],
  "publisher": "Farrar, Straus and Giroux",
  "url": "https://openlibrary.org/search?q=Artificial+Intelligence+A+Guide+for+Thinking+Humans+Mitchell",
  "note": "A clear account of what current AI can and cannot do and why understanding remains hard."
 },
 {
  "id": "dbd9ff32e0",
  "title": "Beijing AI Principles",
  "creators": "Beijing Academy of Artificial Intelligence",
  "year": 2019,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "China",
  "url": "https://www.baai.ac.cn/blog/beijing-ai-principles",
  "note": "Principles for research, use, and governance of AI released by BAAI in May 2019."
 },
 {
  "id": "54a71b9573",
  "title": "Benchmarking Neural Network Robustness to Common Corruptions and Perturbations",
  "creators": "Dan Hendrycks, Thomas Dietterich",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Robustness"
  ],
  "publisher": "ICLR",
  "url": "https://arxiv.org/abs/1903.12261",
  "note": "Introduces ImageNet-C and ImageNet-P."
 },
 {
  "id": "5af9fd6ebe",
  "title": "Captum",
  "creators": "Meta",
  "year": 2019,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/meta-pytorch/captum",
  "note": "Model interpretability and attribution library for PyTorch."
 },
 {
  "id": "a96fc07462",
  "title": "Center for Security and Emerging Technology (CSET)",
  "creators": "Washington, DC",
  "year": 2019,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy",
   "Compute"
  ],
  "publisher": "Academic",
  "url": "https://cset.georgetown.edu",
  "note": "Georgetown University think tank providing policy analysis on AI and national security."
 },
 {
  "id": "7a14aaef4b",
  "title": "Don't Fear the Terminator",
  "creators": "Anthony Zador, Yann LeCun",
  "year": 2019,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Scientific American",
  "url": "https://blogs.scientificamerican.com/observations/dont-fear-the-terminator/",
  "note": "A skeptical argument that AI need not develop drives for dominance."
 },
 {
  "id": "64c03519eb",
  "title": "Embedded Agency",
  "creators": "Abram Demski, Scott Garrabrant",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1902.09469",
  "note": "Describes open problems for agents that are part of the environment they reason about."
 },
 {
  "id": "888f53e2f4",
  "title": "Ethics Guidelines for Trustworthy AI",
  "creators": "EU High-Level Expert Group on AI",
  "year": 2019,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/library/ethics-guidelines-trustworthy-ai",
  "note": "Guidelines setting seven requirements for trustworthy AI that informed the AI Act."
 },
 {
  "id": "83c960dce8",
  "title": "Fine-Tuning Language Models from Human Preferences",
  "creators": "Daniel M. Ziegler, Nisan Stiennon, Jeffrey Wu, Tom B. Brown, Alec Radford, Dario Amodei, Paul Christiano, Geoffrey Irving",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1909.08593",
  "note": "Applies reward learning from human preferences to language model fine-tuning."
 },
 {
  "id": "59b8be58e5",
  "title": "G20 AI Principles",
  "creators": "G20",
  "year": 2019,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Governance"
  ],
  "publisher": "International",
  "url": "https://www.mofa.go.jp/policy/economy/g20_summit/osaka19/pdf/documents/en/annex_08.pdf",
  "note": "Principles drawn from the OECD AI Principles endorsed at the Osaka G20 summit."
 },
 {
  "id": "7e83284bc0",
  "title": "Government AI Readiness Index",
  "creators": "Oxford Insights",
  "year": 2019,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Oxford Insights",
  "url": "https://oxfordinsights.com/ai-readiness/",
  "note": "Annual ranking of how prepared governments are to use AI."
 },
 {
  "id": "5273e9298e",
  "title": "HellaSwag",
  "creators": "Rowan Zellers et al.",
  "year": 2019,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1905.07830",
  "note": "Adversarially filtered commonsense sentence completion benchmark."
 },
 {
  "id": "5ce28e0706",
  "title": "How to Keep Improving When You're Better Than Any Teacher: Iterated Distillation and Amplification",
  "creators": "Robert Miles",
  "year": 2019,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=v9M2Ho9I9Qo",
  "note": "Miles explains Paul Christiano's iterated amplification proposal."
 },
 {
  "id": "f137547fc5",
  "title": "Human Compatible: Artificial Intelligence and the Problem of Control",
  "creators": "Stuart Russell",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Viking",
  "url": "https://en.wikipedia.org/wiki/Human_Compatible",
  "note": "Proposes rebuilding AI around machines that are uncertain about human preferences and learn them from behaviour."
 },
 {
  "id": "bee4654fbe",
  "title": "Interpretable Machine Learning",
  "creators": "Christoph Molnar",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Self-published (online)",
  "url": "https://christophm.github.io/interpretable-ml-book/",
  "note": "A freely available guide to methods for making machine learning models explainable."
 },
 {
  "id": "8fbd5eaf1b",
  "title": "Is AI Safety a Pascal's Mugging?",
  "creators": "Robert Miles",
  "year": 2019,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=JRuNA2eK7w0",
  "note": "Miles argues that AI risk concerns do not rely on tiny probabilities of huge stakes."
 },
 {
  "id": "d8785900cd",
  "title": "Jaan Tallinn: Need, challenges and scenarios of global governance of AI",
  "creators": "Jaan Tallinn",
  "year": 2019,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "Coalition for a Baruch Plan for AI",
  "url": "https://www.youtube.com/watch?v=4_bKfPJsFCA",
  "note": "Tallinn discusses global governance options for AI."
 },
 {
  "id": "aeabf76888",
  "title": "Many Experts Say We Shouldn't Worry About Superintelligent AI. They're Wrong",
  "creators": "Stuart Russell",
  "year": 2019,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "IEEE Spectrum",
  "url": "https://spectrum.ieee.org/many-experts-say-we-shouldnt-worry-about-superintelligent-ai-theyre-wrong",
  "note": "An excerpt from Human Compatible rebutting common dismissals of AI risk."
 },
 {
  "id": "e34ab0b4d1",
  "title": "Model Cards for Model Reporting",
  "creators": "Margaret Mitchell et al.",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "FAT* 2019",
  "url": "https://arxiv.org/abs/1810.03993",
  "note": "Proposes standardised documentation for trained models."
 },
 {
  "id": "00a1e2e94a",
  "title": "Natural Adversarial Examples (ImageNet-A and ImageNet-O)",
  "creators": "Dan Hendrycks et al.",
  "year": 2019,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1907.07174",
  "note": "Real world images that reliably fool image classifiers."
 },
 {
  "id": "f79dd55a4f",
  "title": "OECD AI Principles",
  "creators": "OECD",
  "year": 2019,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "International",
  "url": "https://oecd.ai/en/ai-principles",
  "note": "The first intergovernmental AI standard, adopted in 2019 and updated in May 2024."
 },
 {
  "id": "7a64ca432a",
  "title": "Possible Minds: Twenty-Five Ways of Looking at AI",
  "creators": "John Brockman (ed.)",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Penguin Press",
  "url": "https://openlibrary.org/search?q=Possible+Minds+Twenty-Five+Ways+of+Looking+at+AI",
  "note": "Essays by leading scientists and thinkers responding to Norbert Wiener's warnings about intelligent machines."
 },
 {
  "id": "2208cc512d",
  "title": "Race After Technology: Abolitionist Tools for the New Jim Code",
  "creators": "Ruha Benjamin",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Polity",
  "url": "https://openlibrary.org/search?q=Race+After+Technology+Ruha+Benjamin",
  "note": "Argues that ostensibly neutral technologies can encode and amplify racial hierarchies."
 },
 {
  "id": "d60e1108e3",
  "title": "Rebooting AI: Building Artificial Intelligence We Can Trust",
  "creators": "Gary Marcus, Ernest Davis",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Robustness"
  ],
  "publisher": "Pantheon",
  "url": "https://openlibrary.org/search?q=Rebooting+AI+Marcus+Davis",
  "note": "Argues that deep learning alone cannot produce trustworthy AI and calls for robust, knowledge-based systems."
 },
 {
  "id": "166c6125dc",
  "title": "Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective",
  "creators": "Tom Everitt, Marcus Hutter, Ramana Kumar, Victoria Krakovna",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1908.04734",
  "note": "Uses causal influence diagrams to analyse and design against reward tampering."
 },
 {
  "id": "bcfb80120c",
  "title": "Risks from Learned Optimization",
  "creators": "Evan Hubinger, Chris van Merwijk, Vladimir Mikulik, Joar Skalse, Scott Garrabrant",
  "year": 2019,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/s/r9tYkB2a8Fp4DN8yB",
  "note": "Introduces mesa-optimization, inner alignment and deceptive alignment."
 },
 {
  "id": "49e4afcf66",
  "title": "Risks from Learned Optimization in Advanced Machine Learning Systems",
  "creators": "Evan Hubinger, Chris van Merwijk, Vladimir Mikulik, Joar Skalse, Scott Garrabrant",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1906.01820",
  "note": "Introduces mesa-optimization, inner alignment and deceptive alignment."
 },
 {
  "id": "25fff224f1",
  "title": "Robin Hanson on AI Takeoff Scenarios: AI Go Foom?",
  "creators": "Robin Hanson; Adam Ford",
  "year": 2019,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Science, Technology and the Future",
  "url": "https://www.youtube.com/watch?v=qk3bQrSfUzs",
  "note": "Hanson explains his skepticism of fast AI takeoff."
 },
 {
  "id": "ec4033d019",
  "title": "Safety Gym",
  "creators": "OpenAI",
  "year": 2019,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/openai/safety-gym",
  "note": "Environments for measuring progress on constrained, safe exploration in reinforcement learning."
 },
 {
  "id": "d60e1b27a0",
  "title": "Stanford Institute for Human-Centered AI (HAI)",
  "creators": "Stanford, CA",
  "year": 2019,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "Academic",
  "url": "https://hai.stanford.edu",
  "note": "Stanford institute known for the annual AI Index report and foundation model transparency research."
 },
 {
  "id": "2a1397dd4f",
  "title": "Survival and Flourishing Fund",
  "creators": "San Francisco, CA",
  "year": 2019,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://survivalandflourishing.fund",
  "note": "Grantmaking process funded largely by Jaan Tallinn that supports AI safety organizations."
 },
 {
  "id": "76b0f85c0f",
  "title": "The AI Does Not Hate You: Superintelligence, Rationality and the Race to Save the World",
  "creators": "Tom Chivers",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "Weidenfeld & Nicolson",
  "url": "https://openlibrary.org/search?q=The+AI+Does+Not+Hate+You+Chivers",
  "note": "A journalist's account of the rationalist community and its concerns about AI risk."
 },
 {
  "id": "306af09bbf",
  "title": "The Age of Surveillance Capitalism",
  "creators": "Shoshana Zuboff",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "PublicAffairs",
  "url": "https://en.wikipedia.org/wiki/The_Age_of_Surveillance_Capitalism",
  "note": "Describes how technology firms extract and monetise behavioural data to predict and shape behaviour."
 },
 {
  "id": "d7ca1701f9",
  "title": "The Big Nine: How the Tech Titans and Their Thinking Machines Could Warp Humanity",
  "creators": "Amy Webb",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Forecasting"
  ],
  "publisher": "PublicAffairs",
  "url": "https://openlibrary.org/search?q=The+Big+Nine+Amy+Webb",
  "note": "Scenarios for how nine large technology companies could shape the future of AI."
 },
 {
  "id": "e99992528e",
  "title": "The Bitter Lesson",
  "creators": "Richard Sutton",
  "year": 2019,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Compute",
   "History"
  ],
  "publisher": "incompleteideas.net",
  "url": "http://www.incompleteideas.net/IncIdeas/BitterLesson.html",
  "note": "Argues that general methods that scale with computation have repeatedly beaten hand-built knowledge in AI."
 },
 {
  "id": "b9a1e5724b",
  "title": "The Ethical Algorithm: The Science of Socially Aware Algorithm Design",
  "creators": "Michael Kearns, Aaron Roth",
  "year": 2019,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=The+Ethical+Algorithm+Kearns+Roth",
  "note": "Explains how privacy, fairness and other values can be built into algorithms technically."
 },
 {
  "id": "9f017d7608",
  "title": "The Vulnerable World Hypothesis",
  "creators": "Nick Bostrom",
  "year": 2019,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Global Policy",
  "url": "https://doi.org/10.1111/1758-5899.12718",
  "note": "Argues some technologies could make civilisational devastation the default unless governance changes."
 },
 {
  "id": "5485d6666b",
  "title": "Training AI Without Writing A Reward Function, with Reward Modelling",
  "creators": "Robert Miles",
  "year": 2019,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=PYylPRX6z4Q",
  "note": "An explainer on reward modelling from human feedback."
 },
 {
  "id": "99ff525c8c",
  "title": "What failure looks like",
  "creators": "Paul Christiano",
  "year": 2019,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/HBxe6wdjxK239zajf/what-failure-looks-like",
  "note": "Describes two gradual failure modes: optimizing easy-to-measure proxies and influence-seeking systems."
 },
 {
  "id": "d389b24cde",
  "title": "WinoGrande",
  "creators": "Keisuke Sakaguchi et al.",
  "year": 2019,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1907.10641",
  "note": "Large scale adversarial Winograd schema challenge for commonsense reasoning."
 },
 {
  "id": "43a88c438e",
  "title": "Your Undivided Attention",
  "creators": "Tristan Harris; Aza Raskin",
  "year": 2019,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Center for Humane Technology",
  "url": "https://www.humanetech.com/podcast",
  "note": "The Center for Humane Technology's podcast, which has covered AI risk extensively since 2023."
 },
 {
  "id": "bd7593c000",
  "title": "iHuman (trailer)",
  "creators": "Tonje Hessen Schei",
  "year": 2019,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "IDFA",
  "url": "https://www.youtube.com/watch?v=0qX49Lqo12c",
  "note": "Trailer for the Norwegian documentary on AI, power and surveillance."
 },
 {
  "id": "65e37d0211",
  "title": "AI Alignment Forum",
  "creators": "Lightcone Infrastructure",
  "year": 2018,
  "kind": "Forum",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Interpretability",
   "Control"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/",
  "note": "The main online forum for technical alignment research posts and discussion."
 },
 {
  "id": "d18d373fc3",
  "title": "AI Gridworlds",
  "creators": "Robert Miles",
  "year": 2018,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Evaluations"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=eElfR_BnL5k",
  "note": "A Computerphile discussion of the AI Safety Gridworlds environments."
 },
 {
  "id": "fe9f74ed77",
  "title": "AI Safety Camp",
  "creators": "International",
  "year": 2018,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.aisafety.camp",
  "note": "Remote research program where participants collaborate on AI safety projects."
 },
 {
  "id": "afb196cc51",
  "title": "AI Safety via Debate",
  "creators": "Geoffrey Irving, Paul Christiano, Dario Amodei",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1805.00899",
  "note": "Proposes training agents through debate judged by humans as a route to scalable oversight."
 },
 {
  "id": "ccbd650a3e",
  "title": "AI Superpowers: China, Silicon Valley, and the New World Order",
  "creators": "Kai-Fu Lee",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance"
  ],
  "publisher": "Houghton Mifflin Harcourt",
  "url": "https://en.wikipedia.org/wiki/AI_Superpowers",
  "note": "Compares US and Chinese AI ecosystems and the economic disruption AI may bring."
 },
 {
  "id": "257f7be815",
  "title": "AI Watch",
  "creators": "Issa Rice",
  "year": 2018,
  "kind": "Guide",
  "group": "Learning",
  "topics": [
   "History"
  ],
  "publisher": "aiwatch.issarice.com",
  "url": "https://aiwatch.issarice.com/",
  "note": "A database tracking people and organizations in AI safety."
 },
 {
  "id": "b1f4ad1832",
  "title": "Ada Lovelace Institute",
  "creators": "London, UK",
  "year": 2018,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.adalovelaceinstitute.org",
  "note": "Independent research institute funded by the Nuffield Foundation studying data and AI governance."
 },
 {
  "id": "3b97500567",
  "title": "Adversarial Robustness Toolbox",
  "creators": "Linux Foundation AI and IBM",
  "year": 2018,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/Trusted-AI/adversarial-robustness-toolbox",
  "note": "Library for evaluating and defending ML models against adversarial attacks."
 },
 {
  "id": "223c63149b",
  "title": "Algorithms of Oppression: How Search Engines Reinforce Racism",
  "creators": "Safiya Umoja Noble",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "NYU Press",
  "url": "https://en.wikipedia.org/wiki/Algorithms_of_Oppression",
  "note": "Documents racial and gender bias embedded in commercial search algorithms."
 },
 {
  "id": "e2bd32ee3b",
  "title": "Alignment Newsletter",
  "creators": "Rohin Shah",
  "year": 2018,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "rohinshah.com",
  "url": "https://rohinshah.com/alignment-newsletter/",
  "note": "An archive of weekly summaries of alignment research from 2018 to 2022."
 },
 {
  "id": "54112a757b",
  "title": "Architects of Intelligence: The Truth About AI from the People Building It",
  "creators": "Martin Ford",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "Packt Publishing",
  "url": "https://openlibrary.org/search?q=Architects+of+Intelligence+Martin+Ford",
  "note": "Interviews with prominent AI researchers on progress, timelines and risks."
 },
 {
  "id": "d1e7b33b95",
  "title": "Army of None: Autonomous Weapons and the Future of War",
  "creators": "Paul Scharre",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "W. W. Norton",
  "url": "https://openlibrary.org/search?q=Army+of+None+Scharre",
  "note": "Examines autonomous weapons and the military, ethical and legal stakes of delegating lethal decisions to machines."
 },
 {
  "id": "edf9fb812d",
  "title": "Artificial Intelligence Safety and Security",
  "creators": "Roman V. Yampolskiy (ed.)",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Cybersecurity"
  ],
  "publisher": "Chapman and Hall/CRC",
  "url": "https://openlibrary.org/search?q=Artificial+Intelligence+Safety+and+Security+Yampolskiy",
  "note": "An edited volume collecting technical and philosophical work on AI safety and security."
 },
 {
  "id": "64c6690e17",
  "title": "Artificial Unintelligence: How Computers Misunderstand the World",
  "creators": "Meredith Broussard",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=Artificial+Unintelligence+Broussard",
  "note": "Critiques technochauvinism and the limits of computational solutions to social problems."
 },
 {
  "id": "9d869d4124",
  "title": "Automating Inequality: How High-Tech Tools Profile, Police, and Punish the Poor",
  "creators": "Virginia Eubanks",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "St. Martin's Press",
  "url": "https://en.wikipedia.org/wiki/Automating_Inequality",
  "note": "Investigates automated decision systems in US public services and their impact on poor communities."
 },
 {
  "id": "001829c8e4",
  "title": "CHAI Bibliography",
  "creators": "Center for Human-Compatible AI",
  "year": 2018,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "Center for Human-Compatible AI",
  "url": "https://humancompatible.ai/bibliography",
  "note": "An annotated bibliography of recommended alignment materials."
 },
 {
  "id": "fe780c316d",
  "title": "Centre for the Governance of AI (GovAI)",
  "creators": "Oxford, UK",
  "year": 2018,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.governance.ai",
  "note": "Originated at the Future of Humanity Institute and became an independent nonprofit researching AI governance."
 },
 {
  "id": "233a7b90c4",
  "title": "ChinAI Newsletter",
  "creators": "Jeffrey Ding",
  "year": 2018,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Substack",
  "url": "https://chinai.substack.com/",
  "note": "Translations and analysis of Chinese writing on AI."
 },
 {
  "id": "fd0e4428a9",
  "title": "Clarifying \"AI alignment\"",
  "creators": "Paul Christiano",
  "year": 2018,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/posts/ZeE7EKHTFMBs8eMxn/clarifying-ai-alignment",
  "note": "Defines intent alignment as an AI trying to do what its operator wants."
 },
 {
  "id": "25db227231",
  "title": "Datasheets for Datasets",
  "creators": "Timnit Gebru et al.",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Communications of the ACM",
  "url": "https://arxiv.org/abs/1803.09010",
  "note": "Proposes standard documentation for datasets."
 },
 {
  "id": "9b130de8c8",
  "title": "Do You Trust This Computer? (film site)",
  "creators": "Chris Paine",
  "year": 2018,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Papercut Films",
  "url": "https://doyoutrustthiscomputer.org/",
  "note": "The official site of the 2018 documentary on the dangers of AI."
 },
 {
  "id": "4acd953af6",
  "title": "Do You Trust This Computer? (official trailer)",
  "creators": "Chris Paine",
  "year": 2018,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Film Threat",
  "url": "https://www.youtube.com/watch?v=U7IfqhaAfjA",
  "note": "Trailer for Chris Paine's 2018 documentary on the dangers of AI featuring Elon Musk and Stuart Russell."
 },
 {
  "id": "7e047d50a4",
  "title": "EU Coordinated Plan on Artificial Intelligence",
  "creators": "European Commission",
  "year": 2018,
  "kind": "Policy",
  "group": "Policy and law",
  "topics": [
   "Policy"
  ],
  "publisher": "European Union",
  "url": "https://digital-strategy.ec.europa.eu/en/policies/plan-ai",
  "note": "Plan coordinating member state AI strategies, reviewed in 2021."
 },
 {
  "id": "846776ae34",
  "title": "Gender Bias in Coreference Resolution (WinoBias)",
  "creators": "Jieyu Zhao et al.",
  "year": 2018,
  "kind": "Dataset",
  "group": "Data and tools",
  "topics": [
   "Ethics"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1804.06876",
  "note": "Winograd style sentences for measuring gender bias in coreference systems."
 },
 {
  "id": "a7d2212c8b",
  "title": "Gender Shades: Intersectional Accuracy Disparities in Commercial Gender Classification",
  "creators": "Joy Buolamwini, Timnit Gebru",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Evaluations",
   "Ethics"
  ],
  "publisher": "FAT* 2018 (PMLR)",
  "url": "https://proceedings.mlr.press/v81/buolamwini18a.html",
  "note": "Audits commercial face analysis systems and finds large error disparities by skin type and gender."
 },
 {
  "id": "a4d8b590f9",
  "title": "Google DeepMind Safety Research",
  "creators": "Google DeepMind",
  "year": 2018,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Interpretability"
  ],
  "publisher": "Medium",
  "url": "https://deepmindsafetyresearch.medium.com/",
  "note": "Blog of Google DeepMind's AGI safety and alignment teams."
 },
 {
  "id": "d19f6e9520",
  "title": "GovAI Research",
  "creators": "Centre for the Governance of AI",
  "year": 2018,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "GovAI",
  "url": "https://www.governance.ai/research",
  "note": "Research and analysis on governing advanced AI."
 },
 {
  "id": "575a47870e",
  "title": "Gradient Institute",
  "creators": "Sydney, Australia",
  "year": 2018,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.gradientinstitute.org",
  "note": "Australian research institute working on responsible and safe AI."
 },
 {
  "id": "9ab807c06e",
  "title": "How the Enlightenment Ends",
  "creators": "Henry Kissinger",
  "year": 2018,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "The Atlantic",
  "url": "https://www.theatlantic.com/magazine/archive/2018/06/henry-kissinger-ai-could-mean-the-end-of-human-history/559124/",
  "note": "Kissinger warns that AI could upend human reasoning and political order."
 },
 {
  "id": "eb770d03cb",
  "title": "How to get empowered, not overpowered, by AI",
  "creators": "Max Tegmark",
  "year": 2018,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=2LRwvU6gEbA",
  "note": "Tegmark's earlier TED talk on steering the future of artificial general intelligence."
 },
 {
  "id": "ed9b11dc4b",
  "title": "Intelligence and Stupidity: The Orthogonality Thesis",
  "creators": "Robert Miles",
  "year": 2018,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=hEUO6pjwFOo",
  "note": "An explanation of why intelligence and final goals can vary independently."
 },
 {
  "id": "0ba687fe33",
  "title": "Iterated Amplification",
  "creators": "Paul Christiano et al.",
  "year": 2018,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/s/EmDuGeRw749sD3GKd",
  "note": "A sequence explaining iterated amplification as a scalable oversight approach."
 },
 {
  "id": "ed02d76bbb",
  "title": "Lex Fridman Podcast",
  "creators": "Lex Fridman",
  "year": 2018,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://lexfridman.com/podcast/",
  "note": "A long-form podcast with many episodes featuring AI lab leaders and AI risk researchers."
 },
 {
  "id": "6125da1083",
  "title": "Lucid",
  "creators": "Google Brain and OpenAI Clarity team",
  "year": 2018,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Interpretability"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/tensorflow/lucid",
  "note": "Infrastructure and notebooks for neural network feature visualization."
 },
 {
  "id": "8f337f4f74",
  "title": "Montreal Declaration for a Responsible Development of AI",
  "creators": "Universite de Montreal",
  "year": 2018,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Ethics"
  ],
  "publisher": "Academic",
  "url": "https://montrealdeclaration-responsibleai.com",
  "note": "Declaration of ten ethical principles developed through public deliberation."
 },
 {
  "id": "9da23b4c7d",
  "title": "On the Future: Prospects for Humanity",
  "creators": "Martin Rees",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Biosecurity"
  ],
  "publisher": "Princeton University Press",
  "url": "https://openlibrary.org/search?q=On+the+Future+Prospects+for+Humanity+Rees",
  "note": "Surveys threats and opportunities from biotechnology, AI and climate change."
 },
 {
  "id": "79e3dc981d",
  "title": "Ought / Elicit",
  "creators": "Oakland, CA",
  "year": 2018,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://elicit.com",
  "note": "Ought researched factored cognition and process supervision before spinning out the Elicit research assistant."
 },
 {
  "id": "e343b1f96f",
  "title": "Penalizing Side Effects Using Stepwise Relative Reachability",
  "creators": "Victoria Krakovna, Laurent Orseau, Ramana Kumar, Miljan Martic, Shane Legg",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1806.01186",
  "note": "Proposes an impact measure to discourage agents from causing irreversible side effects."
 },
 {
  "id": "b35bcf4757",
  "title": "Robot Rights",
  "creators": "David J. Gunkel",
  "year": 2018,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=Robot+Rights+Gunkel",
  "note": "Examines whether and how robots and AI systems could have moral or legal rights."
 },
 {
  "id": "a6ccb69817",
  "title": "Safe Exploration: Concrete Problems in AI Safety Part 6",
  "creators": "Robert Miles",
  "year": 2018,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Robustness"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=V527HCWfBCU",
  "note": "Miles explains the safe exploration problem in reinforcement learning."
 },
 {
  "id": "8d9f13fa12",
  "title": "Sanity Checks for Saliency Maps",
  "creators": "Julius Adebayo, Justin Gilmer, Michael Muelly, Ian Goodfellow, Moritz Hardt, Been Kim",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/1810.03292",
  "note": "Shows many saliency methods are insensitive to model parameters."
 },
 {
  "id": "5347df91bd",
  "title": "Scalable Agent Alignment via Reward Modeling: A Research Direction",
  "creators": "Jan Leike, David Krueger, Tom Everitt, Miljan Martic, Vishal Maini, Shane Legg",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1811.07871",
  "note": "Outlines recursive reward modeling as an approach to aligning agents."
 },
 {
  "id": "f38243cf5b",
  "title": "Solving the alignment problem and handing off the future to AI",
  "creators": "Paul Christiano; Rob Wiblin",
  "year": 2018,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment"
  ],
  "publisher": "80,000 Hours",
  "url": "https://www.youtube.com/watch?v=pkIJgZcf-Qo",
  "note": "Christiano's first 80,000 Hours episode describing iterated amplification and prosaic AI alignment."
 },
 {
  "id": "964f8841aa",
  "title": "Specification gaming examples in AI",
  "creators": "Victoria Krakovna",
  "year": 2018,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Victoria Krakovna blog",
  "url": "https://vkrakovna.wordpress.com/2018/04/02/specification-gaming-examples-in-ai/",
  "note": "Introduces the master list of specification gaming examples."
 },
 {
  "id": "06186a5ab2",
  "title": "Stuart Russell: Long-Term Future of Artificial Intelligence, Lex Fridman Podcast #9",
  "creators": "Stuart Russell; Lex Fridman",
  "year": 2018,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Lex Fridman Podcast",
  "url": "https://www.youtube.com/watch?v=KsZI5oXBC0k",
  "note": "An early Lex Fridman episode on the control problem and Russell's approach to beneficial AI."
 },
 {
  "id": "a418e7f63b",
  "title": "Supervising Strong Learners by Amplifying Weak Experts",
  "creators": "Paul Christiano, Buck Shlegeris, Dario Amodei",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1810.08575",
  "note": "Introduces iterated amplification for training systems on tasks humans cannot directly evaluate."
 },
 {
  "id": "786887d751",
  "title": "Takeoff speeds",
  "creators": "Paul Christiano",
  "year": 2018,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "The sideways view",
  "url": "https://sideways-view.com/2018/02/24/takeoff-speeds/",
  "note": "Argues for a slow takeoff in which economic output doubles over years before it doubles faster."
 },
 {
  "id": "277728d46d",
  "title": "The Best of LessWrong",
  "creators": "LessWrong",
  "year": 2018,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Forecasting"
  ],
  "publisher": "LessWrong",
  "url": "https://www.alignmentforum.org/bestoflesswrong",
  "note": "Annual community-reviewed collections of the most valuable posts, many on AI."
 },
 {
  "id": "23fcca9f8b",
  "title": "The Library",
  "creators": "AI Alignment Forum",
  "year": 2018,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/library",
  "note": "Curated sequences of core alignment writing."
 },
 {
  "id": "c9cf5ddaf6",
  "title": "The Malicious Use of Artificial Intelligence: Forecasting, Prevention, and Mitigation",
  "creators": "Miles Brundage et al.",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Cybersecurity"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1802.07228",
  "note": "Surveys digital, physical and political security threats from malicious AI use."
 },
 {
  "id": "2d79871c72",
  "title": "The Moral Machine Experiment",
  "creators": "Edmond Awad et al.",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Ethics"
  ],
  "publisher": "Nature",
  "url": "https://doi.org/10.1038/s41586-018-0637-6",
  "note": "Collects 40 million moral decisions on autonomous vehicle dilemmas across countries."
 },
 {
  "id": "a89bb574d5",
  "title": "The Rocket Alignment Problem",
  "creators": "Eliezer Yudkowsky",
  "year": 2018,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/2018/10/03/rocket-alignment/",
  "note": "A dialogue analogy explaining why MIRI pursues foundational research before building aligned AI."
 },
 {
  "id": "5c8a2c5bda",
  "title": "The Surprising Creativity of Digital Evolution",
  "creators": "Joel Lehman et al.",
  "year": 2018,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1803.03453",
  "note": "Collects anecdotes of evolutionary algorithms exploiting loopholes in their specifications."
 },
 {
  "id": "6164fd10f7",
  "title": "The case for taking AI seriously as a threat to humanity",
  "creators": "Kelsey Piper",
  "year": 2018,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Vox",
  "url": "https://www.vox.com/future-perfect/2018/12/21/18126576/ai-artificial-intelligence-machine-learning-safety-alignment",
  "note": "A widely shared explainer of why AI could pose catastrophic risks."
 },
 {
  "id": "c25ccac6da",
  "title": "Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge",
  "creators": "Allen Institute for AI",
  "year": 2018,
  "kind": "Benchmark",
  "group": "Data and tools",
  "topics": [
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1803.05457",
  "note": "Grade school science questions split into an easy set and a challenge set."
 },
 {
  "id": "df1dbe9857",
  "title": "Toronto Declaration",
  "creators": "Amnesty International and Access Now",
  "year": 2018,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.torontodeclaration.org",
  "note": "Declaration on protecting equality and non-discrimination in machine learning systems."
 },
 {
  "id": "a736f01acf",
  "title": "Value Learning",
  "creators": "Rohin Shah et al.",
  "year": 2018,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "AI Alignment Forum",
  "url": "https://www.alignmentforum.org/s/4dHMdK5TLN6xcqtyc",
  "note": "A sequence on learning human values and its limits."
 },
 {
  "id": "3d9c952149",
  "title": "Why Not Just: Think of AGI Like a Corporation?",
  "creators": "Robert Miles",
  "year": 2018,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=L5pUA3LsEaw",
  "note": "Miles explains why corporations are a poor analogy for AGI."
 },
 {
  "id": "f716d1388f",
  "title": "Why Would AI Want to do Bad Things? Instrumental Convergence",
  "creators": "Robert Miles",
  "year": 2018,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=ZeecOKBus3Q",
  "note": "Miles explains why many goals lead to subgoals like self-preservation and resource acquisition."
 },
 {
  "id": "9762d194ed",
  "title": "3 principles for creating safer AI",
  "creators": "Stuart Russell",
  "year": 2017,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=EBK-a94IFHY",
  "note": "Russell proposes that machines should be uncertain about human preferences and learn them from behaviour."
 },
 {
  "id": "598ae4c2ea",
  "title": "80,000 Hours (YouTube channel)",
  "creators": "80,000 Hours",
  "year": 2017,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@eightythousandhours",
  "note": "Video versions of the 80,000 Hours podcast plus explainer videos on AI risk."
 },
 {
  "id": "a5ac6685a7",
  "title": "AI Index Report",
  "creators": "Stanford HAI",
  "year": 2017,
  "kind": "Report",
  "group": "Policy and law",
  "topics": [
   "Policy",
   "Forecasting"
  ],
  "publisher": "Academic",
  "url": "https://hai.stanford.edu/ai-index",
  "note": "Annual report tracking AI technical progress, investment, and policy."
 },
 {
  "id": "39278fad3c",
  "title": "AI Now Institute",
  "creators": "New York, NY",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://ainowinstitute.org",
  "note": "Policy research institute focused on the social implications of AI and concentration of industry power."
 },
 {
  "id": "1039bfb49f",
  "title": "AI Safety Gridworlds",
  "creators": "Jan Leike, Miljan Martic, Victoria Krakovna, Pedro A. Ortega, Tom Everitt, Andrew Lefrancq, Laurent Orseau, Shane Legg",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1711.09883",
  "note": "A suite of simple environments for testing safety properties of RL agents."
 },
 {
  "id": "36e7100695",
  "title": "AI Stop Button Problem",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=3TYT1QfdfsM",
  "note": "Miles explains corrigibility and why an AI may resist being switched off."
 },
 {
  "id": "6bd0530daf",
  "title": "AI safety resources",
  "creators": "Victoria Krakovna",
  "year": 2017,
  "kind": "Reading list",
  "group": "Learning",
  "topics": [
   "Alignment"
  ],
  "publisher": "Victoria Krakovna blog",
  "url": "https://vkrakovna.wordpress.com/ai-safety-resources/",
  "note": "A long-maintained list of introductory AI safety readings, research agendas and courses."
 },
 {
  "id": "868ced32b3",
  "title": "AlphaGo (film site)",
  "creators": "Greg Kohs",
  "year": 2017,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Moxie Pictures",
  "url": "https://www.alphagomovie.com/",
  "note": "The official site of the AlphaGo documentary."
 },
 {
  "id": "06eaf8aeb7",
  "title": "AlphaGo: The Movie (full documentary)",
  "creators": "Greg Kohs",
  "year": 2017,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Google DeepMind",
  "url": "https://www.youtube.com/watch?v=WXuK6gekU1Y",
  "note": "The documentary of AlphaGo's 2016 match against Lee Sedol."
 },
 {
  "id": "dd973085f9",
  "title": "Asilomar AI Principles",
  "creators": "Future of Life Institute",
  "year": 2017,
  "kind": "Framework",
  "group": "Policy and law",
  "topics": [
   "Existential risk",
   "Ethics",
   "History"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org/open-letter/ai-principles/",
  "note": "Twenty-three principles developed at the 2017 Beneficial AI conference."
 },
 {
  "id": "f3e427d2fd",
  "title": "Avoiding Negative Side Effects: Concrete Problems in AI Safety part 1",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=lqJUIqZNzP8",
  "note": "The first video in Miles's Concrete Problems series, on side effects."
 },
 {
  "id": "92ffc49bff",
  "title": "Axiomatic Attribution for Deep Networks",
  "creators": "Mukund Sundararajan, Ankur Taly, Qiqi Yan",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "ICML",
  "url": "https://arxiv.org/abs/1703.01365",
  "note": "Introduces integrated gradients for feature attribution."
 },
 {
  "id": "3a269c2ac0",
  "title": "Berkeley Existential Risk Initiative (BERI)",
  "creators": "Berkeley, CA",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://existence.org",
  "note": "Nonprofit that provides operational support to university research groups working on existential risk."
 },
 {
  "id": "b13c820839",
  "title": "Deep Reinforcement Learning from Human Preferences",
  "creators": "Paul Christiano, Jan Leike, Tom B. Brown, Miljan Martic, Shane Legg, Dario Amodei",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/1706.03741",
  "note": "Trains agents from human comparisons of trajectories, the basis of RLHF."
 },
 {
  "id": "6d288a0563",
  "title": "Don't Worry About the Vase (WordPress archive)",
  "creators": "Zvi Mowshowitz",
  "year": 2017,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Policy",
   "Forecasting"
  ],
  "publisher": "WordPress",
  "url": "https://thezvi.wordpress.com/",
  "note": "The mirror and archive of Zvi Mowshowitz's blog posts."
 },
 {
  "id": "e7337ef25b",
  "title": "Feature Visualization",
  "creators": "Chris Olah, Alexander Mordvintsev, Ludwig Schubert",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "Distill",
  "url": "https://distill.pub/2017/feature-visualization/",
  "note": "Explains optimisation-based visualisation of what neurons respond to."
 },
 {
  "id": "a46c192cc3",
  "title": "General AI Won't Want You To Fix its Code",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=4l7Is6vOAOA",
  "note": "Miles explains goal preservation as a convergent incentive."
 },
 {
  "id": "e7ad6eeb79",
  "title": "Inverse Reward Design",
  "creators": "Dylan Hadfield-Menell, Smitha Milli, Pieter Abbeel, Stuart Russell, Anca Dragan",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/1711.02827",
  "note": "Treats a designed reward as evidence about intended behaviour rather than as the true objective."
 },
 {
  "id": "d0cb22eb51",
  "title": "Katja Grace: Empirical evidence on the future of AI",
  "creators": "Katja Grace",
  "year": 2017,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Forecasting"
  ],
  "publisher": "GoCAS Existential risk",
  "url": "https://www.youtube.com/watch?v=i3gS1yTws3Y",
  "note": "Grace presents early AI Impacts survey and forecasting work."
 },
 {
  "id": "ed7f55df78",
  "title": "Life 3.0: Being Human in the Age of Artificial Intelligence",
  "creators": "Max Tegmark",
  "year": 2017,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Alfred A. Knopf",
  "url": "https://en.wikipedia.org/wiki/Life_3.0",
  "note": "Surveys possible futures with advanced AI and argues for steering the transition deliberately."
 },
 {
  "id": "e6d9487bd4",
  "title": "Lightcone Infrastructure",
  "creators": "Berkeley, CA",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.lightconeinfrastructure.com",
  "note": "Organization that runs LessWrong and the AI Alignment Forum."
 },
 {
  "id": "fa9d726c7c",
  "title": "Lil'Log",
  "creators": "Lilian Weng",
  "year": 2017,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "lilianweng.github.io",
  "url": "https://lilianweng.github.io/",
  "note": "Detailed survey posts on ML topics including reward hacking and adversarial attacks."
 },
 {
  "id": "bef7372bd5",
  "title": "Long-Term Future Fund",
  "creators": "International",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://funds.effectivealtruism.org/funds/far-future",
  "note": "EA Funds grantmaker that supports independent AI safety researchers and small projects."
 },
 {
  "id": "81b307ce9f",
  "title": "Microsoft Responsible AI",
  "creators": "Redmond, WA",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Company",
  "url": "https://www.microsoft.com/en-us/ai/responsible-ai",
  "note": "Microsoft's responsible AI program, which publishes a Responsible AI Standard and transparency reports."
 },
 {
  "id": "434fe52d07",
  "title": "Network Dissection: Quantifying Interpretability of Deep Visual Representations",
  "creators": "David Bau, Bolei Zhou, Aditya Khosla, Aude Oliva, Antonio Torralba",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "CVPR",
  "url": "https://arxiv.org/abs/1704.05796",
  "note": "Measures how individual units align with human-interpretable concepts."
 },
 {
  "id": "ef4c4ce841",
  "title": "Open Philanthropy (Coefficient Giving)",
  "creators": "San Francisco, CA",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Nonprofit",
  "url": "https://coefficientgiving.org",
  "note": "Major funder of AI safety and governance research, renamed Coefficient Giving in 2025."
 },
 {
  "id": "b9534257a4",
  "title": "Provably Beneficial AI",
  "creators": "Stuart Russell",
  "year": 2017,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=Kw_1N9Nfir0",
  "note": "Russell's talk at the Beneficial AI conference organized by the Future of Life Institute."
 },
 {
  "id": "0e20bdbdf2",
  "title": "Reward Hacking: Concrete Problems in AI Safety Part 3",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=92qDfT8pENs",
  "note": "Part of Miles's series walking through the Concrete Problems in AI Safety paper."
 },
 {
  "id": "0e7416cb58",
  "title": "Robert Miles AI Safety",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@RobertMilesAI",
  "note": "The leading explainer channel for AI alignment concepts."
 },
 {
  "id": "7124e4f22c",
  "title": "Robot Ethics 2.0: From Autonomous Cars to Artificial Intelligence",
  "creators": "Patrick Lin, Ryan Jenkins, Keith Abney (eds.)",
  "year": 2017,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=Robot+Ethics+2.0",
  "note": "Essays on the ethics of autonomous systems from cars to weapons to AI."
 },
 {
  "id": "9ad7d6969c",
  "title": "Slaughterbots",
  "creators": "Future of Life Institute; Stuart Russell",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=HipTO_7mUOw",
  "note": "A short dystopian film about autonomous weapons made to support a ban."
 },
 {
  "id": "9ec7de1849",
  "title": "Stop Button Solution?",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Control"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=9nktr1MgS-A",
  "note": "A follow-up on cooperative inverse reinforcement learning as a proposed solution to the stop button problem."
 },
 {
  "id": "9bc96da605",
  "title": "Superintelligence: Science or Fiction?",
  "creators": "Elon Musk; Stuart Russell; Ray Kurzweil; Demis Hassabis; Sam Harris; Nick Bostrom; David Chalmers; Bart Selman; Jaan Tallinn",
  "year": 2017,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Future of Life Institute",
  "url": "https://www.youtube.com/watch?v=OFBwz4R6Fi0",
  "note": "A panel at the 2017 Beneficial AI conference in Asilomar."
 },
 {
  "id": "ca4dd86b52",
  "title": "The 80,000 Hours Podcast",
  "creators": "Rob Wiblin; Luisa Rodriguez",
  "year": 2017,
  "kind": "Podcast",
  "group": "Podcasts",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk"
  ],
  "publisher": "80,000 Hours",
  "url": "https://80000hours.org/podcast/",
  "note": "Long-form interviews with researchers on AI safety, governance and other pressing problems."
 },
 {
  "id": "771f2c910e",
  "title": "The Off-Switch Game",
  "creators": "Dylan Hadfield-Menell, Anca Dragan, Pieter Abbeel, Stuart Russell",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "IJCAI",
  "url": "https://arxiv.org/abs/1611.08219",
  "note": "Shows that uncertainty about the human's objective gives an agent an incentive to allow shutdown."
 },
 {
  "id": "f449b9e3bb",
  "title": "The Seven Deadly Sins of AI Predictions",
  "creators": "Rodney Brooks",
  "year": 2017,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Forecasting"
  ],
  "publisher": "MIT Technology Review",
  "url": "https://www.technologyreview.com/2017/10/06/241837/the-seven-deadly-sins-of-ai-predictions/",
  "note": "A roboticist's critique of common errors in forecasting AI."
 },
 {
  "id": "33c21566b2",
  "title": "There's No Fire Alarm for Artificial General Intelligence",
  "creators": "Eliezer Yudkowsky",
  "year": 2017,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/2017/10/13/fire-alarm/",
  "note": "Argues there will be no clear signal telling society that AGI is imminent."
 },
 {
  "id": "1f24ac8984",
  "title": "Towards A Rigorous Science of Interpretable Machine Learning",
  "creators": "Finale Doshi-Velez, Been Kim",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Interpretability"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1702.08608",
  "note": "Proposes a taxonomy for evaluating interpretability."
 },
 {
  "id": "8fdfe2d732",
  "title": "Towards Deep Learning Models Resistant to Adversarial Attacks",
  "creators": "Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, Adrian Vladu",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "ICLR 2018",
  "url": "https://arxiv.org/abs/1706.06083",
  "note": "Frames robustness as robust optimisation and introduces PGD adversarial training."
 },
 {
  "id": "cba9190101",
  "title": "Vector Institute",
  "creators": "Toronto, Canada",
  "year": 2017,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://vectorinstitute.ai",
  "note": "Canadian AI research institute with work on trustworthy AI."
 },
 {
  "id": "253eaebd2e",
  "title": "When Will AI Exceed Human Performance? Evidence from AI Experts",
  "creators": "Katja Grace, John Salvatier, Allan Dafoe, Baobao Zhang, Owain Evans",
  "year": 2017,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Journal of Artificial Intelligence Research",
  "url": "https://arxiv.org/abs/1705.08807",
  "note": "Surveys ML researchers on timelines for human-level AI."
 },
 {
  "id": "0f93f4f1a7",
  "title": "Why Not Just: Raise AI Like Kids?",
  "creators": "Robert Miles",
  "year": 2017,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment"
  ],
  "publisher": "Robert Miles AI Safety (YouTube)",
  "url": "https://www.youtube.com/watch?v=eaYIU6YXr3w",
  "note": "Miles explains why raising AI like a child would not reliably produce good values."
 },
 {
  "id": "697f724b34",
  "title": "A Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks",
  "creators": "Dan Hendrycks, Kevin Gimpel",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "ICLR 2017",
  "url": "https://arxiv.org/abs/1610.02136",
  "note": "Establishes the maximum softmax probability baseline for OOD detection."
 },
 {
  "id": "10ed2d9580",
  "title": "AI Alignment: Why It's Hard, and Where to Start",
  "creators": "Eliezer Yudkowsky",
  "year": 2016,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/stanford-talk/",
  "note": "An edited talk introducing core alignment difficulties such as corrigibility and value specification."
 },
 {
  "id": "e8e08aa269",
  "title": "Alignment for Advanced Machine Learning Systems",
  "creators": "Jessica Taylor, Eliezer Yudkowsky, Patrick LaVictoire, Andrew Critch",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/files/AlignmentMachineLearning.pdf",
  "note": "A MIRI research agenda for aligning machine learning based systems."
 },
 {
  "id": "962077734f",
  "title": "Can we build AI without losing control over it?",
  "creators": "Sam Harris",
  "year": 2016,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Control",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=8nt3edWLgIg",
  "note": "Harris argues that we are unprepared for the arrival of superintelligent AI."
 },
 {
  "id": "928bc3e2a0",
  "title": "Center for Human-Compatible AI (CHAI)",
  "creators": "Berkeley, CA",
  "year": 2016,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment"
  ],
  "publisher": "Academic",
  "url": "https://humancompatible.ai",
  "note": "UC Berkeley research center led by Stuart Russell that studies provably beneficial AI and assistance games."
 },
 {
  "id": "65be26c1b0",
  "title": "CleverHans",
  "creators": "CleverHans team",
  "year": 2016,
  "kind": "Tool",
  "group": "Data and tools",
  "topics": [
   "Robustness"
  ],
  "publisher": "GitHub",
  "url": "https://github.com/cleverhans-lab/cleverhans",
  "note": "Early adversarial example library for benchmarking robustness."
 },
 {
  "id": "03f589573f",
  "title": "Concrete Problems in AI Safety",
  "creators": "Dario Amodei, Chris Olah, Jacob Steinhardt, Paul Christiano, John Schulman, Dan Mané",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1606.06565",
  "note": "Identifies five practical research problems including side effects, reward hacking and safe exploration."
 },
 {
  "id": "284828fb21",
  "title": "Concrete Problems in AI Safety (Paper)",
  "creators": "Robert Miles",
  "year": 2016,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Robustness"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=AjyM-f8rDpg",
  "note": "Miles introduces the 2016 Concrete Problems in AI Safety paper."
 },
 {
  "id": "a186582263",
  "title": "Cooperative Inverse Reinforcement Learning",
  "creators": "Dylan Hadfield-Menell, Anca Dragan, Pieter Abbeel, Stuart Russell",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "NeurIPS",
  "url": "https://arxiv.org/abs/1606.03137",
  "note": "Frames alignment as a cooperative game where a robot learns a human's reward."
 },
 {
  "id": "8fbbc5bfa4",
  "title": "Dylan Hadfield-Menell: The Off-Switch",
  "creators": "Dylan Hadfield-Menell",
  "year": 2016,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Control"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://www.youtube.com/watch?v=t06IciZknDg",
  "note": "A talk on the off-switch game at the MIRI CSRBAI workshop."
 },
 {
  "id": "18c5917c9c",
  "title": "How I'm fighting bias in algorithms",
  "creators": "Joy Buolamwini",
  "year": 2016,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Ethics"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=UG_X_7g63rY",
  "note": "Buolamwini describes the coded gaze and her work on bias in facial recognition."
 },
 {
  "id": "f4044819e0",
  "title": "Import AI",
  "creators": "Jack Clark",
  "year": 2016,
  "kind": "Newsletter",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Policy",
   "Forecasting"
  ],
  "publisher": "jack-clark.net",
  "url": "https://jack-clark.net/",
  "note": "A weekly newsletter on AI research and policy by Anthropic cofounder Jack Clark."
 },
 {
  "id": "5517a7e731",
  "title": "Leverhulme Centre for the Future of Intelligence",
  "creators": "Cambridge, UK",
  "year": 2016,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Existential risk",
   "Ethics"
  ],
  "publisher": "Academic",
  "url": "https://www.lcfi.ac.uk",
  "note": "Interdisciplinary University of Cambridge centre studying the opportunities and risks of AI."
 },
 {
  "id": "265c233b7a",
  "title": "Lo and Behold: Reveries of the Connected World (official trailer)",
  "creators": "Werner Herzog",
  "year": 2016,
  "kind": "Documentary",
  "group": "TV and video",
  "topics": [
   "History"
  ],
  "publisher": "Magnolia Pictures",
  "url": "https://www.youtube.com/watch?v=Zc1tZ8JsZvg",
  "note": "Trailer for Herzog's documentary on the internet and AI."
 },
 {
  "id": "8638aede4c",
  "title": "Logical Induction",
  "creators": "Scott Garrabrant, Tsvi Benson-Tilsen, Andrew Critch, Nate Soares, Jessica Taylor",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment"
  ],
  "publisher": "arXiv",
  "url": "https://arxiv.org/abs/1609.03543",
  "note": "Gives a computable algorithm for assigning coherent probabilities to logical statements over time."
 },
 {
  "id": "928a8da362",
  "title": "Partnership on AI",
  "creators": "San Francisco, CA",
  "year": 2016,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://partnershiponai.org",
  "note": "Multistakeholder coalition of companies and civil society that publishes guidance on safe foundation model deployment."
 },
 {
  "id": "92d2af2f89",
  "title": "Racing to the Precipice: A Model of Artificial Intelligence Development",
  "creators": "Stuart Armstrong, Nick Bostrom, Carl Shulman",
  "year": 2016,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "AI & Society",
  "url": "https://doi.org/10.1007/s00146-015-0590-y",
  "note": "Models how competitive races can push developers to skimp on safety."
 },
 {
  "id": "2903796a51",
  "title": "Sam Altman's Manifest Destiny",
  "creators": "Tad Friend",
  "year": 2016,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "History"
  ],
  "publisher": "The New Yorker",
  "url": "https://www.newyorker.com/magazine/2016/10/10/sam-altmans-manifest-destiny",
  "note": "A profile of Sam Altman and the early ambitions of OpenAI."
 },
 {
  "id": "68e5382747",
  "title": "Superintelligence FAQ",
  "creators": "Scott Alexander",
  "year": 2016,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/posts/LTtNXM9shNM9AC2mp/superintelligence-faq",
  "note": "An accessible question-and-answer introduction to superintelligence risk."
 },
 {
  "id": "74495cab75",
  "title": "The Age of Em: Work, Love, and Life When Robots Rule the Earth",
  "creators": "Robin Hanson",
  "year": 2016,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Oxford University Press",
  "url": "https://en.wikipedia.org/wiki/The_Age_of_Em",
  "note": "Forecasts a world dominated by whole-brain emulations using economics and social science."
 },
 {
  "id": "f857c19b13",
  "title": "Tony Blair Institute for Global Change (technology policy)",
  "creators": "London, UK",
  "year": 2016,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://institute.global",
  "note": "Policy institute that has published proposals on state capacity for AI safety."
 },
 {
  "id": "30bb51ae8f",
  "title": "Weapons of Math Destruction",
  "creators": "Cathy O'Neil",
  "year": 2016,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics"
  ],
  "publisher": "Crown",
  "url": "https://en.wikipedia.org/wiki/Weapons_of_Math_Destruction",
  "note": "Shows how opaque, large-scale algorithms can entrench inequality and harm people at scale."
 },
 {
  "id": "fd5271a0e4",
  "title": "Why Tool AIs Want to Be Agent AIs",
  "creators": "Gwern Branwen",
  "year": 2016,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Agents"
  ],
  "publisher": "gwern.net",
  "url": "https://gwern.net/tool-ai",
  "note": "Argues economic incentives push AI systems from passive tools toward autonomous agents."
 },
 {
  "id": "ff9c70563e",
  "title": "AI Self Improvement",
  "creators": "Robert Miles",
  "year": 2015,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=5qfIgCiYlfY",
  "note": "An early Computerphile video on recursive self-improvement."
 },
 {
  "id": "04cdd42292",
  "title": "Autonomous Weapons Open Letter",
  "creators": "Future of Life Institute",
  "year": 2015,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Ethics",
   "History"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org/open-letter/open-letter-autonomous-weapons-ai-robotics/",
  "note": "Letter calling for a ban on offensive autonomous weapons beyond meaningful human control."
 },
 {
  "id": "5f1c09cc48",
  "title": "Corrigibility",
  "creators": "Nate Soares, Benja Fallenstein, Eliezer Yudkowsky, Stuart Armstrong",
  "year": 2015,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Control"
  ],
  "publisher": "AAAI Workshop on AI and Ethics",
  "url": "https://intelligence.org/files/Corrigibility.pdf",
  "note": "Formalises the problem of building agents that do not resist correction or shutdown."
 },
 {
  "id": "260343a916",
  "title": "Deadly Truth of General AI?",
  "creators": "Robert Miles",
  "year": 2015,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Computerphile",
  "url": "https://www.youtube.com/watch?v=tcdVC4e6EV4",
  "note": "Miles's first Computerphile appearance, introducing the risks of general AI."
 },
 {
  "id": "2948f26100",
  "title": "Future of Life Institute (YouTube channel)",
  "creators": "Future of Life Institute",
  "year": 2015,
  "kind": "Channel",
  "group": "TV and video",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "YouTube",
  "url": "https://www.youtube.com/@FutureofLifeInstitute",
  "note": "FLI's channel with podcast episodes, short films and conference talks."
 },
 {
  "id": "c85368bf58",
  "title": "Metaculus AI questions",
  "creators": "Metaculus",
  "year": 2015,
  "kind": "Tracker",
  "group": "Data and tools",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Metaculus",
  "url": "https://www.metaculus.com/questions/?categories=artificial-intelligence",
  "note": "Community forecasts on AI capabilities, timelines, policy, and risk."
 },
 {
  "id": "94a846ec68",
  "title": "OpenAI Safety",
  "creators": "San Francisco, CA",
  "year": 2015,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Evaluations"
  ],
  "publisher": "Company",
  "url": "https://openai.com/safety",
  "note": "OpenAI publishes its safety approach, system cards, and preparedness evaluations on this page."
 },
 {
  "id": "09c8125a7f",
  "title": "Rationality: From AI to Zombies",
  "creators": "Eliezer Yudkowsky",
  "year": 2015,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://www.readthesequences.com",
  "note": "A compilation of the Sequences on reasoning and decision making, including foundational essays on AI risk."
 },
 {
  "id": "736347338c",
  "title": "Rationality: From AI to Zombies (The Sequences)",
  "creators": "Eliezer Yudkowsky",
  "year": 2015,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/rationality",
  "note": "The collected Sequences on rationality that seeded the LessWrong and AI alignment communities."
 },
 {
  "id": "498434e8be",
  "title": "Research Priorities for Robust and Beneficial Artificial Intelligence",
  "creators": "Stuart Russell, Daniel Dewey, Max Tegmark",
  "year": 2015,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "AI Magazine",
  "url": "https://arxiv.org/abs/1602.03506",
  "note": "The research agenda attached to the 2015 open letter on beneficial AI."
 },
 {
  "id": "7fe73a9a7a",
  "title": "Research Priorities for Robust and Beneficial Artificial Intelligence: An Open Letter",
  "creators": "Future of Life Institute",
  "year": 2015,
  "kind": "Statement",
  "group": "Policy and law",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org/open-letter/ai-open-letter/",
  "note": "2015 letter signed by thousands of researchers calling for research on making AI beneficial."
 },
 {
  "id": "81b9d84a10",
  "title": "The AI Revolution: Our Immortality or Extinction",
  "creators": "Tim Urban",
  "year": 2015,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Wait But Why",
  "url": "https://waitbutwhy.com/2015/01/artificial-intelligence-revolution-2.html",
  "note": "The second part of the series on why superintelligence could be either the best or the last thing to happen to humanity."
 },
 {
  "id": "8c75b55d17",
  "title": "The AI Revolution: The Road to Superintelligence",
  "creators": "Tim Urban",
  "year": 2015,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Wait But Why",
  "url": "https://waitbutwhy.com/2015/01/artificial-intelligence-revolution-1.html",
  "note": "A popular illustrated introduction to exponential progress and the path from narrow AI to superintelligence."
 },
 {
  "id": "4cf74af2d7",
  "title": "The Doomsday Invention",
  "creators": "Raffi Khatchadourian",
  "year": 2015,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "The New Yorker",
  "url": "https://www.newyorker.com/magazine/2015/11/23/doomsday-invention-artificial-intelligence-nick-bostrom",
  "note": "A profile of Nick Bostrom and the early superintelligence risk movement."
 },
 {
  "id": "f6c63c1847",
  "title": "The Technological Singularity",
  "creators": "Murray Shanahan",
  "year": 2015,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=The+Technological+Singularity+Shanahan",
  "note": "A concise MIT Press Essential Knowledge introduction to the idea of a singularity and its risks."
 },
 {
  "id": "96047865b9",
  "title": "What happens when our computers get smarter than we are?",
  "creators": "Nick Bostrom",
  "year": 2015,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "TED",
  "url": "https://www.youtube.com/watch?v=MnT1xgZgkpk",
  "note": "Bostrom's TED talk on the control problem for superintelligent machines."
 },
 {
  "id": "b0b31271a7",
  "title": "AI Impacts",
  "creators": "Berkeley, CA",
  "year": 2014,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "Nonprofit",
  "url": "https://aiimpacts.org",
  "note": "Research project known for its surveys of AI researchers on timelines and risk."
 },
 {
  "id": "5840395ce1",
  "title": "Effective Altruism Forum: AI safety topic",
  "creators": "Centre for Effective Altruism",
  "year": 2014,
  "kind": "Forum",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "EA Forum",
  "url": "https://forum.effectivealtruism.org/topics/ai-safety",
  "note": "Forum posts on AI safety strategy, careers and funding."
 },
 {
  "id": "3429e4d51a",
  "title": "Explaining and Harnessing Adversarial Examples",
  "creators": "Ian J. Goodfellow, Jonathon Shlens, Christian Szegedy",
  "year": 2014,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "ICLR 2015",
  "url": "https://arxiv.org/abs/1412.6572",
  "note": "Attributes adversarial examples to linearity and introduces the fast gradient sign method."
 },
 {
  "id": "b10046068c",
  "title": "Future of Life Institute (FLI)",
  "creators": "Campbell, CA",
  "year": 2014,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://futureoflife.org",
  "note": "Nonprofit that organized the 2023 pause letter and publishes the AI Safety Index of company practices."
 },
 {
  "id": "adcb3f138b",
  "title": "Humans Need Not Apply (now titled What Happened to Horses Is Happening to Us)",
  "creators": "CGP Grey",
  "year": 2014,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Governance"
  ],
  "publisher": "CGP Grey",
  "url": "https://www.youtube.com/watch?v=7Pq-S557XQU",
  "note": "An influential video on automation and the displacement of human labour."
 },
 {
  "id": "6462aa3966",
  "title": "Meditations on Moloch",
  "creators": "Scott Alexander",
  "year": 2014,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Slate Star Codex",
  "url": "https://slatestarcodex.com/2014/07/30/meditations-on-moloch/",
  "note": "An essay on multipolar traps and competitive dynamics that destroy shared values, widely cited in AI governance."
 },
 {
  "id": "36a983d4ea",
  "title": "Smarter Than Us: The Rise of Machine Intelligence",
  "creators": "Stuart Armstrong",
  "year": 2014,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://openlibrary.org/search?q=Smarter+Than+Us+Stuart+Armstrong",
  "note": "A short introduction to why specifying safe goals for powerful AI is difficult."
 },
 {
  "id": "4dfae42dee",
  "title": "Stephen Hawking: AI could spell end of the human race",
  "creators": "Stephen Hawking; Rory Cellan-Jones",
  "year": 2014,
  "kind": "TV",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "BBC News",
  "url": "https://www.youtube.com/watch?v=fFLVyWBDTfo",
  "note": "Hawking's 2014 BBC interview warning that full AI could end the human race."
 },
 {
  "id": "1a2221799f",
  "title": "Superintelligence",
  "creators": "Nick Bostrom",
  "year": 2014,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Talks at Google",
  "url": "https://www.youtube.com/watch?v=pywF6ZzsghI",
  "note": "Bostrom presents his book Superintelligence at Google."
 },
 {
  "id": "f156e51acc",
  "title": "Superintelligence: Paths, Dangers, Strategies",
  "creators": "Nick Bostrom",
  "year": 2014,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Oxford University Press",
  "url": "https://en.wikipedia.org/wiki/Superintelligence:_Paths,_Dangers,_Strategies",
  "note": "Argues that a machine superintelligence could gain a decisive strategic advantage and that controlling it is a hard, unsolved problem."
 },
 {
  "id": "f20d68086c",
  "title": "Tesla's Elon Musk: We're Summoning the Demon with Artificial Intelligence",
  "creators": "Elon Musk",
  "year": 2014,
  "kind": "Video",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "Bloomberg Originals",
  "url": "https://www.youtube.com/watch?v=Tzb_CSRO-0g",
  "note": "Musk's 2014 MIT remark comparing AI development to summoning a demon."
 },
 {
  "id": "97776dd04f",
  "title": "The Future Society",
  "creators": "Boston, MA",
  "year": 2014,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://thefuturesociety.org",
  "note": "Nonprofit that has worked on EU AI Act implementation and international AI governance."
 },
 {
  "id": "c86570a6a9",
  "title": "Center on Long-Term Risk",
  "creators": "London, UK",
  "year": 2013,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://longtermrisk.org",
  "note": "Research group focused on reducing risks of astronomical suffering, including from AI conflict."
 },
 {
  "id": "82829efb37",
  "title": "Intriguing Properties of Neural Networks",
  "creators": "Christian Szegedy et al.",
  "year": 2013,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Robustness"
  ],
  "publisher": "ICLR 2014",
  "url": "https://arxiv.org/abs/1312.6199",
  "note": "Discovers adversarial examples in neural networks."
 },
 {
  "id": "45acece6cc",
  "title": "Meta FAIR responsible AI",
  "creators": "Menlo Park, CA",
  "year": 2013,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Robustness",
   "Ethics"
  ],
  "publisher": "Company",
  "url": "https://ai.meta.com/responsible-ai/",
  "note": "Meta's AI research division, which publishes safety tooling such as Llama Guard."
 },
 {
  "id": "83f72cb07c",
  "title": "Our Final Invention: Artificial Intelligence and the End of the Human Era",
  "creators": "James Barrat",
  "year": 2013,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Thomas Dunne Books",
  "url": "https://en.wikipedia.org/wiki/Our_Final_Invention",
  "note": "A journalistic account warning that artificial general intelligence could be humanity's last invention."
 },
 {
  "id": "25e1276099",
  "title": "The Hanson-Yudkowsky AI-Foom Debate",
  "creators": "Robin Hanson, Eliezer Yudkowsky",
  "year": 2013,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/ai-foom-debate/",
  "note": "Collected blog debate from 2008 on whether AI would undergo a rapid local intelligence explosion."
 },
 {
  "id": "b2d18544e4",
  "title": "Centre for the Study of Existential Risk (CSER)",
  "creators": "Cambridge, UK",
  "year": 2012,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Existential risk"
  ],
  "publisher": "Academic",
  "url": "https://www.cser.ac.uk",
  "note": "University of Cambridge centre studying extreme technological risks including those from AI."
 },
 {
  "id": "f6cf7d4f0a",
  "title": "The Superintelligent Will: Motivation and Instrumental Rationality in Advanced Artificial Agents",
  "creators": "Nick Bostrom",
  "year": 2012,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Minds and Machines",
  "url": "https://nickbostrom.com/superintelligentwill.pdf",
  "note": "States the orthogonality and instrumental convergence theses."
 },
 {
  "id": "78e64a558e",
  "title": "Yudkowsky vs Hanson: Singularity Debate",
  "creators": "Eliezer Yudkowsky; Robin Hanson",
  "year": 2011,
  "kind": "Talk",
  "group": "TV and video",
  "topics": [
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "Jane Street",
  "url": "https://www.youtube.com/watch?v=TuXl-iidnFY",
  "note": "A live version of the AI foom debate about the speed and locality of an intelligence explosion."
 },
 {
  "id": "0e84423bce",
  "title": "Google DeepMind Safety and Responsibility",
  "creators": "London, UK",
  "year": 2010,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Interpretability",
   "Evaluations"
  ],
  "publisher": "Company",
  "url": "https://deepmind.google/about/responsibility-safety/",
  "note": "DeepMind's safety and alignment teams work on frontier safety, interpretability, and amplified oversight."
 },
 {
  "id": "0ed23d8d67",
  "title": "gwern.net",
  "creators": "Gwern Branwen",
  "year": 2010,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Compute",
   "Forecasting"
  ],
  "publisher": "gwern.net",
  "url": "https://gwern.net/",
  "note": "Gwern's essays, including influential writing on scaling and AI risk."
 },
 {
  "id": "d40d6849d8",
  "title": "LessWrong",
  "creators": "Lightcone Infrastructure",
  "year": 2009,
  "kind": "Forum",
  "group": "Learning",
  "topics": [
   "Alignment",
   "Existential risk",
   "Forecasting"
  ],
  "publisher": "LessWrong",
  "url": "https://www.lesswrong.com/",
  "note": "A community blog on rationality and AI that hosts much early and ongoing AI safety writing."
 },
 {
  "id": "8c7931f84e",
  "title": "The Quest for Artificial Intelligence: A History of Ideas and Achievements",
  "creators": "Nils J. Nilsson",
  "year": 2009,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "History"
  ],
  "publisher": "Cambridge University Press",
  "url": "https://ai.stanford.edu/~nilsson/QAI/qai.pdf",
  "note": "A comprehensive history of AI research by one of its pioneers."
 },
 {
  "id": "2d52fd162a",
  "title": "Artificial Intelligence as a Positive and Negative Factor in Global Risk",
  "creators": "Eliezer Yudkowsky",
  "year": 2008,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "Global Catastrophic Risks (Oxford University Press)",
  "url": "https://intelligence.org/files/AIPosNegFactor.pdf",
  "note": "An early statement of the case that unfriendly AI is a global catastrophic risk."
 },
 {
  "id": "8b65966d27",
  "title": "Global Catastrophic Risks",
  "creators": "Nick Bostrom, Milan M. Ćirković (eds.)",
  "year": 2008,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Oxford University Press",
  "url": "https://en.wikipedia.org/wiki/Global_Catastrophic_Risks_(book)",
  "note": "An edited volume on catastrophic risks including Yudkowsky's chapter on AI as a factor in global risk."
 },
 {
  "id": "7eb9716d96",
  "title": "Moral Machines: Teaching Robots Right from Wrong",
  "creators": "Wendell Wallach, Colin Allen",
  "year": 2008,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "Ethics"
  ],
  "publisher": "Oxford University Press",
  "url": "https://openlibrary.org/search?q=Moral+Machines+Wallach+Allen",
  "note": "An early treatment of machine ethics and how artificial agents might make moral decisions."
 },
 {
  "id": "c904f6cae5",
  "title": "The Basic AI Drives",
  "creators": "Stephen M. Omohundro",
  "year": 2008,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "Existential risk"
  ],
  "publisher": "AGI 2008",
  "url": "https://selfawaresystems.com/wp-content/uploads/2008/01/ai_drives_final.pdf",
  "note": "Argues that sufficiently advanced goal-seeking systems will tend to acquire resources and resist shutdown."
 },
 {
  "id": "144b2183d7",
  "title": "Center for a New American Security (Technology and National Security)",
  "creators": "Washington, DC",
  "year": 2007,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Compute"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.cnas.org/research/technology-and-national-security",
  "note": "Think tank with research on AI safety, compute governance, and great power competition."
 },
 {
  "id": "e1a7bcc75b",
  "title": "Machine Intelligence Research Institute Blog",
  "creators": "MIRI",
  "year": 2007,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment",
   "Policy"
  ],
  "publisher": "Machine Intelligence Research Institute",
  "url": "https://intelligence.org/blog/",
  "note": "Updates and essays from MIRI, now focused on policy to halt dangerous AI development."
 },
 {
  "id": "fee626bc25",
  "title": "Future of Humanity Institute (FHI)",
  "creators": "Oxford, UK",
  "year": 2005,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Existential risk",
   "History"
  ],
  "publisher": "Academic",
  "url": "https://www.futureofhumanityinstitute.org",
  "note": "Oxford institute founded by Nick Bostrom that shaped early AI risk research and closed in April 2024."
 },
 {
  "id": "9e2aee910c",
  "title": "Shtetl-Optimized",
  "creators": "Scott Aaronson",
  "year": 2005,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Alignment"
  ],
  "publisher": "scottaaronson.blog",
  "url": "https://scottaaronson.blog/",
  "note": "A theoretical computer scientist's blog, with posts from his time working on safety at OpenAI."
 },
 {
  "id": "4499ef5075",
  "title": "The Singularity Is Near: When Humans Transcend Biology",
  "creators": "Ray Kurzweil",
  "year": 2005,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Forecasting"
  ],
  "publisher": "Viking",
  "url": "https://en.wikipedia.org/wiki/The_Singularity_Is_Near",
  "note": "Forecasts exponential technological progress leading to a technological singularity."
 },
 {
  "id": "d3853ad302",
  "title": "Ethical Issues in Advanced Artificial Intelligence",
  "creators": "Nick Bostrom",
  "year": 2003,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "Ethics",
   "History"
  ],
  "publisher": "nickbostrom.com",
  "url": "https://nickbostrom.com/ethics/ai",
  "note": "An early essay on the ethics and risks of superintelligence, including the paperclip example."
 },
 {
  "id": "87bace736c",
  "title": "Our Final Hour: A Scientist's Warning",
  "creators": "Martin Rees",
  "year": 2003,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Basic Books",
  "url": "https://en.wikipedia.org/wiki/Our_Final_Hour",
  "note": "Argues that humanity has perhaps a 50 percent chance of surviving the century given emerging technologies."
 },
 {
  "id": "9c0fa08178",
  "title": "Existential Risks: Analyzing Human Extinction Scenarios and Related Hazards",
  "creators": "Nick Bostrom",
  "year": 2002,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk"
  ],
  "publisher": "Journal of Evolution and Technology",
  "url": "https://nickbostrom.com/existential/risks",
  "note": "Defines existential risk and catalogues scenarios including badly programmed superintelligence."
 },
 {
  "id": "17bd221a97",
  "title": "Simon Willison's Weblog",
  "creators": "Simon Willison",
  "year": 2002,
  "kind": "Blog",
  "group": "Writing",
  "topics": [
   "Agents",
   "Cybersecurity"
  ],
  "publisher": "simonwillison.net",
  "url": "https://simonwillison.net/",
  "note": "Running commentary on LLMs with sustained coverage of prompt injection."
 },
 {
  "id": "3bb54653ee",
  "title": "Nuclear Threat Initiative (biosecurity and AI)",
  "creators": "Washington, DC",
  "year": 2001,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Biosecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.nti.org/area/biological/",
  "note": "The Nuclear Threat Initiative works on AI and biosecurity risk reduction."
 },
 {
  "id": "f900388407",
  "title": "Machine Intelligence Research Institute (MIRI)",
  "creators": "Berkeley, CA",
  "year": 2000,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Governance",
   "Existential risk"
  ],
  "publisher": "Nonprofit",
  "url": "https://intelligence.org",
  "note": "Founded as the Singularity Institute, it pioneered agent foundations research and now focuses on policy advocacy to halt frontier AI development."
 },
 {
  "id": "3435743b4e",
  "title": "Why the Future Doesn't Need Us",
  "creators": "Bill Joy",
  "year": 2000,
  "kind": "Article",
  "group": "Writing",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "Wired",
  "url": "https://www.wired.com/2000/04/joy-2/",
  "note": "A Sun Microsystems cofounder warns about robotics, genetic engineering and nanotech."
 },
 {
  "id": "9e48506341",
  "title": "How Long Before Superintelligence?",
  "creators": "Nick Bostrom",
  "year": 1998,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "nickbostrom.com",
  "url": "https://nickbostrom.com/superintelligence",
  "note": "An early forecast of when superintelligence might arrive based on hardware and software trends."
 },
 {
  "id": "39ec5269ec",
  "title": "Center for Democracy and Technology (AI Governance Lab)",
  "creators": "Washington, DC",
  "year": 1994,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Ethics"
  ],
  "publisher": "Nonprofit",
  "url": "https://cdt.org",
  "note": "Civil liberties organization whose AI Governance Lab works on practical AI accountability."
 },
 {
  "id": "fd41a41f14",
  "title": "Mila AI safety research",
  "creators": "Montreal, Canada",
  "year": 1993,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Alignment",
   "Governance"
  ],
  "publisher": "Academic",
  "url": "https://mila.quebec",
  "note": "Quebec AI institute founded by Yoshua Bengio, whose AI safety work supports the International AI Safety Report."
 },
 {
  "id": "20ca4032be",
  "title": "The Coming Technological Singularity",
  "creators": "Vernor Vinge",
  "year": 1993,
  "kind": "Essay",
  "group": "Writing",
  "topics": [
   "Forecasting",
   "History"
  ],
  "publisher": "San Diego State University",
  "url": "https://edoras.sdsu.edu/~vinge/misc/singularity.html",
  "note": "The classic essay predicting superhuman intelligence and the end of the human era."
 },
 {
  "id": "d6ebc5351a",
  "title": "Machines Who Think",
  "creators": "Pamela McCorduck",
  "year": 1979,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "History"
  ],
  "publisher": "W. H. Freeman",
  "url": "https://openlibrary.org/search?q=Machines+Who+Think+McCorduck",
  "note": "An early history of artificial intelligence and its founders."
 },
 {
  "id": "34902cc270",
  "title": "Computer Power and Human Reason: From Judgment to Calculation",
  "creators": "Joseph Weizenbaum",
  "year": 1976,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics",
   "History"
  ],
  "publisher": "W. H. Freeman",
  "url": "https://en.wikipedia.org/wiki/Computer_Power_and_Human_Reason",
  "note": "The ELIZA creator argues that some decisions should never be delegated to computers."
 },
 {
  "id": "bf8e0ea8a0",
  "title": "Speculations Concerning the First Ultraintelligent Machine",
  "creators": "I. J. Good",
  "year": 1965,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Existential risk",
   "History"
  ],
  "publisher": "Advances in Computers",
  "url": "https://doi.org/10.1016/S0065-2458(08)60418-0",
  "note": "Introduces the idea of an intelligence explosion triggered by an ultraintelligent machine."
 },
 {
  "id": "b2a160d9be",
  "title": "God and Golem, Inc.: A Comment on Certain Points Where Cybernetics Impinges on Religion",
  "creators": "Norbert Wiener",
  "year": 1964,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "MIT Press",
  "url": "https://openlibrary.org/search?q=God+and+Golem+Inc+Wiener",
  "note": "Discusses learning machines and the danger of machines pursuing literally specified goals."
 },
 {
  "id": "b45d15428b",
  "title": "Some Moral and Technical Consequences of Automation",
  "creators": "Norbert Wiener",
  "year": 1960,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "Alignment",
   "History"
  ],
  "publisher": "Science",
  "url": "https://doi.org/10.1126/science.131.3410.1355",
  "note": "Warns that we must be sure the purpose put into a machine is the purpose we really desire."
 },
 {
  "id": "17a25bae19",
  "title": "Computing Machinery and Intelligence",
  "creators": "Alan M. Turing",
  "year": 1950,
  "kind": "Paper",
  "group": "Papers",
  "topics": [
   "History"
  ],
  "publisher": "Mind",
  "url": "https://doi.org/10.1093/mind/LIX.236.433",
  "note": "Introduces the imitation game and anticipates objections to machine intelligence."
 },
 {
  "id": "44dca7a983",
  "title": "The Human Use of Human Beings: Cybernetics and Society",
  "creators": "Norbert Wiener",
  "year": 1950,
  "kind": "Book",
  "group": "Books",
  "topics": [
   "Ethics",
   "History"
  ],
  "publisher": "Houghton Mifflin",
  "url": "https://en.wikipedia.org/wiki/The_Human_Use_of_Human_Beings",
  "note": "An early warning about the social consequences of automation and machine control."
 },
 {
  "id": "ce3365c584",
  "title": "RAND Corporation (AI policy research)",
  "creators": "Santa Monica, CA",
  "year": 1948,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy",
   "Biosecurity",
   "Cybersecurity"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.rand.org/topics/artificial-intelligence.html",
  "note": "Research organization that has published influential reports on securing model weights and AI and biological risk."
 },
 {
  "id": "64c76ba68b",
  "title": "Brookings AI governance research",
  "creators": "Washington, DC",
  "year": 1916,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://www.brookings.edu/topics/artificial-intelligence/",
  "note": "Think tank publishing AI policy analysis for US policymakers."
 },
 {
  "id": "bcfa23d399",
  "title": "Carnegie Endowment AI program",
  "creators": "Washington, DC",
  "year": 1910,
  "kind": "Organization",
  "group": "Organizations",
  "topics": [
   "Governance",
   "Policy"
  ],
  "publisher": "Nonprofit",
  "url": "https://carnegieendowment.org/programs/technology-and-international-affairs",
  "note": "Think tank program publishing research on international AI governance."
 }
]