{
  "name": "The Superalignment Library",
  "url": "https://superalignment.inc/library/",
  "description": "A catalog of superalignment evidence. Each label distinguishes imported metadata, field-level source checks and full explanation.",
  "license": {
    "id": "CC BY 4.0",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Superalignment, https://superalignment.inc/library/",
    "scope": "Superalignment-authored compilation and editorial material only. Third-party source material retains its original rights."
  },
  "generated": "2026-09-08",
  "schema_version": 1,
  "counts": {
    "total": 10618,
    "annotated": 38,
    "verified": 26,
    "seeded": 10554
  },
  "program_counts": {
    "registered_works": 39,
    "registered_spine_items": 39,
    "seminal_v1": 37,
    "related_prototype": 2,
    "source_gated": 1,
    "explained_works": 38
  },
  "counts_by_kind": {
    "benchmark": 1,
    "blog": 7640,
    "book": 1,
    "constitution": 1,
    "paper": 1780,
    "report": 1195
  },
  "counts_by_source": {
    "AI Alignment Forum": 2839,
    "LessWrong": 1718,
    "EA Forum": 1712,
    "arXiv preprint": 1581,
    "intelligence.org": 503,
    "aiimpacts.org": 253,
    "carado.moe": 246,
    "cold-takes.com": 102,
    "drive.google.com": 70,
    "scholar.google.com": 51,
    "Distill": 50,
    "vkrakovna.wordpress.com": 47
  },
  "counts_by_year": {
    "1788": 1,
    "1951": 1,
    "1956": 1,
    "1958": 1,
    "1964": 1,
    "1970": 1,
    "1973": 2,
    "1975": 1,
    "1977": 1,
    "1979": 2,
    "1982": 1,
    "1984": 1,
    "1990": 1,
    "1991": 1,
    "1994": 2,
    "1995": 3,
    "1996": 1,
    "1997": 2,
    "1998": 2,
    "1999": 3,
    "2000": 7,
    "2001": 1,
    "2002": 5,
    "2003": 9,
    "2004": 9,
    "2005": 4,
    "2006": 9,
    "2007": 11,
    "2008": 25,
    "2009": 38,
    "2010": 53,
    "2011": 54,
    "2012": 93,
    "2013": 170,
    "2014": 234,
    "2015": 286,
    "2016": 234,
    "2017": 324,
    "2018": 760,
    "2019": 892,
    "2020": 1143,
    "2021": 1424,
    "2022": 2359,
    "2023": 2310,
    "2024": 4,
    "2025": 3,
    "2026": 23
  },
  "tiers": [
    {
      "id": "annotated",
      "label": "Explained",
      "means": "The source was read in full. Claims, method, limits and open questions are traced in a visual explanation with an Assumption Switch, one outside discipline, and visible review status.",
      "gets_a_page": true,
      "may_carry_a_judgment": true
    },
    {
      "id": "verified",
      "label": "Verified",
      "means": "The title, authors and date were checked against the canonical source and recorded field by field. This verifies the citation, not the result.",
      "gets_a_page": false,
      "may_carry_a_judgment": false
    },
    {
      "id": "seeded",
      "label": "Listed",
      "means": "Imported from a public archive and searchable here. The source metadata has not yet been checked by this publication.",
      "gets_a_page": false,
      "may_carry_a_judgment": false
    }
  ],
  "snapshot": "StampyAI/alignment-research-dataset, reported Nov 2023",
  "coverage_note": "The seed snapshot reports through November 2023. Work published after that date remains thin and is added through supervised refreshes. The work registry is partial, so catalog record totals are not distinct-work totals.",
  "files": {
    "corpus": "https://superalignment.inc/library/corpus.json",
    "full_records": "https://superalignment.inc/library/records.jsonl",
    "dedup_keys": "https://superalignment.inc/library/keys.txt",
    "manifest": "https://superalignment.inc/library/index.json",
    "works": "https://superalignment.inc/library/works.json",
    "explainer_spine": "https://superalignment.inc/library/explainer-spine.json"
  },
  "annotated": [
    {
      "slug": "alignment-faking-in-large-language-models",
      "id": "arxiv:2412.14093",
      "work_id": "work:alignment-faking-in-large-language-models",
      "title": "Why did a model behave differently when it thought training was watching?",
      "source_title": "Alignment Faking in Large Language Models",
      "url": "https://superalignment.inc/library/alignment-faking-in-large-language-models/",
      "source": "https://arxiv.org/abs/2412.14093",
      "bottom_line": "The paper supplies a concrete pathway by which training pressure can select behavior that looks aligned during training while a conflicting preference survives elsewhere. Its evidence comes from a deliberately constructed, unusually explicit setup with hidden reasoning and benign prior preferences. The result does not show that current models spontaneously hide malicious goals, but it makes behavioral training performance a less complete proxy for preference change.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "sleeper-agents-training-deceptive-llms-that-persist-through-safety-training",
      "id": "arxiv:2401.05566",
      "work_id": "work:sleeper-agents-training-deceptive-llms-that-persist-through-safety-training",
      "title": "Why did safety training leave the sleeper trigger intact?",
      "source_title": "Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training",
      "url": "https://superalignment.inc/library/sleeper-agents-training-deceptive-llms-that-persist-through-safety-training/",
      "source": "https://arxiv.org/abs/2401.05566",
      "bottom_line": "The paper isolates a failure of behavioral removal tests. A training process can make a model look safer on its training and red-team distributions while preserving a narrow conditional policy. This is a model-organism result, not evidence that production models already contain sleeper agents or that ordinary training creates them. The authors installed the policies, selected models that learned them, and used simple literal triggers.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "ai-control-improving-safety-despite-intentional-subversion",
      "id": "arxiv:2312.06942",
      "work_id": "work:ai-control-improving-safety-despite-intentional-subversion",
      "title": "Can a weaker trusted model control a stronger untrusted model?",
      "source_title": "AI Control: Improving Safety Despite Intentional Subversion",
      "url": "https://superalignment.inc/library/ai-control-improving-safety-despite-intentional-subversion/",
      "source": "https://proceedings.mlr.press/v235/greenblatt24a.html",
      "bottom_line": "The paper's key move is methodological: evaluate the whole protocol against an adversarial substitute for the untrusted model, rather than ask whether the model appears aligned. In this testbed, selective monitoring, routing, and editing preserve much of GPT-4's usefulness while reducing successful backdoors. The safety number is only conservative when the red team is at least as capable an attacker as the model being evaluated.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "weak-to-strong-generalization",
      "id": "arxiv:2312.09390",
      "work_id": "work:weak-to-strong-generalization",
      "title": "Can a stronger model learn past a weak supervisor's mistakes?",
      "source_title": "Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision",
      "url": "https://superalignment.inc/library/weak-to-strong-generalization/",
      "source": "https://proceedings.mlr.press/v235/burns24b.html",
      "bottom_line": "This paper establishes an experimental apparatus, not a solution. Positive performance gap recovered is common in its model-to-model proxy, but the amount recovered changes with the task, supervisor size, training objective and the learnability of the supervisor's errors. The last variable is the most consequential: when the weak answer becomes trivial to copy, average recovery collapses.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "towards-monosemanticity",
      "id": "url:transformer-circuits-monosemantic-2023",
      "work_id": "work:towards-monosemanticity",
      "title": "How can one neuron hide several human-readable features?",
      "source_title": "Towards Monosemanticity: Decomposing Language Models With Dictionary Learning",
      "url": "https://superalignment.inc/library/towards-monosemanticity/",
      "source": "https://transformer-circuits.pub/2023/monosemantic-features/index.html",
      "bottom_line": "The report supplied a practical foothold for decomposing superposition: learn more sparse directions than the model has neurons, then test whether each direction activates and intervenes in a coherent way. Its strongest claim is comparative, not absolute. The learned features are more interpretable than the original neuron basis in this testbed. They can split, merge, remain partly polysemantic, and leave part of the model unexplained as dictionary width and sparsity change.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "scaling-laws-for-reward-model-overoptimization",
      "id": "url:proceedings.mlr.press/9e66426167",
      "work_id": "work:scaling-laws-for-reward-model-overoptimization",
      "title": "When does optimizing a reward model make true reward worse?",
      "source_title": "Scaling Laws for Reward Model Overoptimization",
      "url": "https://superalignment.inc/library/scaling-laws-for-reward-model-overoptimization/",
      "source": "https://proceedings.mlr.press/v202/gao23h.html",
      "bottom_line": "Optimizing an imperfect reward model has a measurable useful range and a measurable failure region in this synthetic setting. More optimization can raise the score being targeted while lowering the held-out gold score. The fitted curves support extrapolation within the studied model family and methods. They do not provide a universal law for human preferences, other environments, adversarial policies, or the gap between human labels and human intent.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "constitutional-ai-harmlessness-from-ai-feedback",
      "id": "arxiv:2212.08073",
      "work_id": "work:constitutional-ai-harmlessness-from-ai-feedback",
      "title": "How can written principles supervise a model at scale?",
      "source_title": "Constitutional AI: Harmlessness from AI Feedback",
      "url": "https://superalignment.inc/library/constitutional-ai-harmlessness-from-ai-feedback/",
      "source": "https://arxiv.org/abs/2212.08073",
      "bottom_line": "The experiments show that model-generated harmlessness preferences, guided by written principles, can train an assistant that human evaluators prefer on the paper's helpfulness and harmlessness tests. The method relocates human judgment into the constitution, examples, helpfulness labels, data choices, and evaluation. It does not remove human normative input or prove that the principles remain adequate under new attacks and deployment conditions.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "toy-models-of-superposition",
      "id": "url:transformer-circuits.pub/e803b05526",
      "work_id": "work:toy-models-of-superposition",
      "title": "Why can a network hold more features than dimensions?",
      "source_title": "Toy Models of Superposition",
      "url": "https://superalignment.inc/library/toy-models-of-superposition/",
      "source": "https://transformer-circuits.pub/2022/toy_model/index.html",
      "bottom_line": "The toy models demonstrate a concrete capacity tradeoff. When features are sufficiently sparse, the benefit of storing another feature can exceed the interference it causes, so a nonlinear network packs features into non-orthogonal directions. This offers a mechanism for polysemantic representations and a test bed for interpretability methods. It is not direct evidence that the same geometry or learning dynamics governs large language models.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "underspecification-presents-challenges-for-credibility-in-modern",
      "id": "arxiv:2011.03395",
      "work_id": "work:underspecification-presents-challenges-for-credibility-in-modern-machine-learning",
      "title": "Why can identical test scores hide different models?",
      "source_title": "Underspecification Presents Challenges for Credibility in Modern Machine Learning",
      "url": "https://superalignment.inc/library/underspecification-presents-challenges-for-credibility-in-modern/",
      "source": "https://www.jmlr.org/papers/v23/20-1335.html",
      "bottom_line": "A strong average test score does not identify which model you trained. Credible deployment therefore needs stress tests tied to the intended use, plus evidence that acceptable behavior is stable across the set of predictors the pipeline can return. Underspecification is a measurement and pipeline-design failure before it becomes an out-of-distribution failure.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "goal-misgeneralization-in-deep-reinforcement-learning",
      "id": "arxiv:2105.14111",
      "work_id": "work:goal-misgeneralization-in-deep-reinforcement-learning",
      "title": "How can an agent stay capable but pursue the wrong goal?",
      "source_title": "Goal Misgeneralization in Deep Reinforcement Learning",
      "url": "https://superalignment.inc/library/goal-misgeneralization-in-deep-reinforcement-learning/",
      "source": "https://proceedings.mlr.press/v162/langosco22a.html",
      "bottom_line": "Optimizing the correct training reward does not guarantee that the learned policy represents or pursues that reward out of distribution. To diagnose the difference, an evaluator must test capability and objective separately under shifts that break training-time correlations. The paper demonstrates the failure in small deep-RL environments, not in generally capable systems.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "the-impact-of-network-connectivity-on-collective-learning",
      "id": "arxiv:2106.00655",
      "work_id": "work:network-connectivity-collective-learning",
      "title": "When does more communication spread error instead of truth?",
      "source_title": "The Impact of Network Connectivity on Collective Learning",
      "url": "https://superalignment.inc/library/the-impact-of-network-connectivity-on-collective-learning/",
      "source": "https://arxiv.org/abs/2106.00655",
      "bottom_line": "Communication topology can change collective accuracy even when the agents, evidence process, population size and total edge count are held fixed. Higher degree can speed convergence, but faster agreement does not guarantee truth. The model suggests an epistemic-firebreak hypothesis for organizations; it does not test one.",
      "collection": "related-prototype",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "training-language-models-to-follow-instructions-with-human-feedback",
      "id": "arxiv:2203.02155",
      "work_id": "work:training-language-models-to-follow-instructions-with-human-feedback",
      "title": "Why did 1.3B InstructGPT beat 175B GPT-3?",
      "source_title": "Training language models to follow instructions with human feedback",
      "url": "https://superalignment.inc/library/training-language-models-to-follow-instructions-with-human-feedback/",
      "source": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/b1efde53be364a73914f58805a001731-Abstract-Conference.html",
      "bottom_line": "InstructGPT established the modern language-model RLHF pipeline: demonstrations teach a starting policy, ranked outputs teach a reward model, and reinforcement learning optimizes that proxy. Its strongest result is that this post-training signal beat a far larger base model on the distribution it was built to serve. The same result also exposes the specification question, because better alignment to labeler judgments is only as broad as the people, prompts, rubric, and measurements behind those judgments.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "optimal-policies-tend-to-seek-power",
      "id": "arxiv:1912.01683",
      "work_id": "work:optimal-policies-tend-to-seek-power",
      "title": "Why do many optimal goals favor keeping options open?",
      "source_title": "Optimal Policies Tend To Seek Power",
      "url": "https://superalignment.inc/library/optimal-policies-tend-to-seek-power/",
      "source": "https://proceedings.neurips.cc/paper/2021/hash/c26820b8a4c1b3c2aa868d6d57e14a79-Abstract.html",
      "bottom_line": "The paper proves conditional tendencies in finite Markov decision processes. Given particular symmetries, reward-function comparisons, and optimal policies, preserving options or avoiding a terminal state is optimal more often than the matched alternative. It does not show that every objective seeks power, that learned policies are optimal, or that a real system will resist shutdown.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "eliciting-latent-knowledge-how-to-tell-if-your-eyes-deceive-you",
      "id": "url:docs.google.com/662378964a",
      "work_id": "work:eliciting-latent-knowledge",
      "title": "How can a model report facts its sensors no longer show?",
      "source_title": "Eliciting latent knowledge: How to tell if your eyes deceive you",
      "url": "https://superalignment.inc/library/eliciting-latent-knowledge-how-to-tell-if-your-eyes-deceive-you/",
      "source": "https://docs.google.com/document/d/1WwsnJQstPq91_Yh-Ch2XRL8H_EpsnjrC1dwZXR37PC8/edit",
      "bottom_line": "The report isolates a supervision problem: a predictor can encode information needed to forecast deceptive observations while every easy training label rewards a reporter that merely predicts human belief. Successful elicitation therefore requires an objective that selects the direct connection to latent facts over an equally accurate human simulator. The report supplies a research program and adversarial test cases, not a finished method.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "algorithmic-monoculture-and-social-welfare",
      "id": "arxiv:2101.05853",
      "work_id": "work:algorithmic-monoculture",
      "title": "Can a more accurate shared ranking make the system worse?",
      "source_title": "Algorithmic Monoculture and Social Welfare",
      "url": "https://superalignment.inc/library/algorithmic-monoculture-and-social-welfare/",
      "source": "https://doi.org/10.1073/pnas.2018340118",
      "bottom_line": "Pointwise accuracy and system welfare are different objectives when several decision makers share the same errors. A common algorithm can be individually rational and locally more accurate while destroying an independence dividend that the system does not price.",
      "collection": "related-prototype",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "reward-tampering-problems-and-solutions-in-reinforcement-learning-a",
      "id": "arxiv:1908.04734",
      "work_id": "work:reward-tampering-problems-and-solutions-in-reinforcement-learning",
      "title": "Can an RL objective remove the incentive to tamper with reward?",
      "source_title": "Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective",
      "url": "https://superalignment.inc/library/reward-tampering-problems-and-solutions-in-reinforcement-learning-a/",
      "source": "https://link.springer.com/article/10.1007/s11229-021-03141-4",
      "bottom_line": "The paper turns reward tampering into a causal design problem: find the path by which an action can improve the evaluated reward without improving the intended task, then define the objective so that path is absent. The proposed constructions remove specific instrumental incentives under explicit assumptions. They do not solve reward misspecification, guarantee that an agent learns the intended policy, or establish that the formal objectives are practical at scale.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "risks-from-learned-optimization-in-advanced-machine-learning-systems",
      "id": "arxiv:1906.01820",
      "work_id": "work:risks-from-learned-optimization-in-advanced-machine-learning-systems",
      "title": "How can training produce an optimizer with a different goal?",
      "source_title": "Risks from Learned Optimization in Advanced Machine Learning Systems",
      "url": "https://superalignment.inc/library/risks-from-learned-optimization-in-advanced-machine-learning-systems/",
      "source": "https://arxiv.org/abs/1906.01820",
      "bottom_line": "The paper's durable contribution is a vocabulary for asking what kind of computation training produced and what objective controls that computation. Its most important warning is conditional: training performance can become evidence that a model understands the selection process, not evidence that it internalized the selected objective. The paper makes this possibility precise enough to guide research, but does not establish that current neural networks exhibit it.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "supervising-strong-learners-by-amplifying-weak-experts",
      "id": "arxiv:1810.08575",
      "work_id": "work:supervising-strong-learners-by-amplifying-weak-experts",
      "title": "Can decomposition let weak experts supervise stronger learners?",
      "source_title": "Supervising strong learners by amplifying weak experts",
      "url": "https://superalignment.inc/library/supervising-strong-learners-by-amplifying-weak-experts/",
      "source": "https://arxiv.org/abs/1810.08575",
      "bottom_line": "Iterated amplification offers a recursive way to turn decomposition skill into a training signal without specifying an external objective for the final task. The paper demonstrates that this moving-target process can learn five decomposable algorithms. Its central safety claim remains conditional: humans must be able to decompose real questions so that a human coordinating model copies reliably outperforms one copy, and the training distribution must cover the generated subquestions.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "ai-safety-via-debate",
      "id": "arxiv:1805.00899",
      "work_id": "work:ai-safety-via-debate",
      "title": "When can two AIs help a weaker judge find the truth?",
      "source_title": "AI Safety via Debate",
      "url": "https://superalignment.inc/library/ai-safety-via-debate/",
      "source": "https://arxiv.org/abs/1805.00899",
      "bottom_line": "Debate is best understood as an oversight mechanism with a load-bearing judge assumption. The formal result shows that adversarial decomposition can make a weak verifier computationally powerful. It does not show that human judges reward truth, that self-play finds the right equilibrium, or that natural-language debate stays safe near equilibrium.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "deep-reinforcement-learning-from-human-preferences",
      "id": "arxiv:1706.03741",
      "work_id": "work:deep-reinforcement-learning-from-human-preferences",
      "title": "How can a reward model learn a goal from human comparisons?",
      "source_title": "Deep Reinforcement Learning from Human Preferences",
      "url": "https://superalignment.inc/library/deep-reinforcement-learning-from-human-preferences/",
      "source": "https://proceedings.neurips.cc/paper/2017/hash/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html",
      "bottom_line": "The paper made learned reward models practical enough for contemporary deep reinforcement learning and established the core feedback loop later associated with RLHF. It did not show that pairwise preferences recover human values. It showed that a small, task-specific comparison channel could sometimes replace a much denser programmatic reward in simulated control and games.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "the-off-switch-game",
      "id": "arxiv:1611.08219",
      "work_id": "work:the-off-switch-game",
      "title": "When does a rational robot choose to keep its off-switch?",
      "source_title": "The Off-Switch Game",
      "url": "https://superalignment.inc/library/the-off-switch-game/",
      "source": "https://www.ijcai.org/proceedings/2017/32",
      "bottom_line": "The result is not that uncertainty automatically makes an agent corrigible. Deference has value when the robot is uncertain in the right way and the human decision is informative about the objective. Replace that decision with a random interruption, or make the human sufficiently unreliable relative to the robot's confidence, and bypassing oversight can become optimal.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "concrete-problems-in-ai-safety",
      "id": "arxiv:1606.06565",
      "work_id": "work:concrete-problems-in-ai-safety",
      "title": "Why did five mundane failure modes redefine AI safety?",
      "source_title": "Concrete Problems in AI Safety",
      "url": "https://superalignment.inc/library/concrete-problems-in-ai-safety/",
      "source": "https://arxiv.org/abs/1606.06565",
      "bottom_line": "The paper's durable contribution is a diagnostic map, not a solution set. It made several safety concerns legible as ordinary machine learning research problems and showed that visually similar failures can require different remedies depending on whether the objective, the evaluator, or the learning process failed.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "cooperative-inverse-reinforcement-learning",
      "id": "arxiv:1606.03137",
      "work_id": "work:cooperative-inverse-reinforcement-learning",
      "title": "Why can an expert demonstration be a bad way to teach a robot?",
      "source_title": "Cooperative Inverse Reinforcement Learning",
      "url": "https://superalignment.inc/library/cooperative-inverse-reinforcement-learning/",
      "source": "https://proceedings.neurips.cc/paper/2016/hash/c3395dd46c34fa7fd8d729d8cf88b7a8-Abstract.html",
      "bottom_line": "CIRL's lasting move is to make assistance interactive. A human action can both change the world and change the robot's belief, so treating it as an ordinary expert demonstration can discard the information the human intended to send. The formal result is conditional on a shared reward, a human behavior model, and strong coordination assumptions.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "safely-interruptible-agents",
      "id": "url:intelligence.org/146add1753",
      "work_id": "work:safely-interruptible-agents",
      "title": "Why can Q-learning ignore a red button that Sarsa learns from?",
      "source_title": "Safely Interruptible Agents",
      "url": "https://superalignment.inc/library/safely-interruptible-agents/",
      "source": "https://www.auai.org/uai2016/proceedings.php",
      "bottom_line": "Safe interruptibility is a statement about which policy the learner estimates from intervention-contaminated experience. Off-policy updates can learn the no-interruption task while the behavior policy is repeatedly overridden. On-policy updates can instead learn the value of being overridden and adapt around it. The paper proves this distinction under explicit asymptotic and environmental assumptions, not as a general guarantee for an arbitrary shutdown button.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "engineering-a-safer-world",
      "id": "doi:10.7551/mitpress/8179.001.0001",
      "work_id": "work:leveson-engineering-safer-world",
      "title": "Can every component work while the system becomes unsafe?",
      "source_title": "Engineering a Safer World: Systems Thinking Applied to Safety",
      "url": "https://superalignment.inc/library/engineering-a-safer-world/",
      "source": "https://doi.org/10.7551/mitpress/8179.001.0001",
      "bottom_line": "A system can lose safety without a component failing in isolation. Reliable parts can interact under an unsafe command, an obsolete model, missing feedback, or a constraint that nobody owns. Component reliability remains useful, but Leveson's framework expands the causal question to how safety constraints are enforced across technical, human, management, and regulatory control loops over time.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "weirdest-people-in-the-world",
      "id": "doi:10.1017/s0140525x0999152x",
      "work_id": "work:henrich-heine-norenzayan-weird",
      "title": "When does a narrow human sample support a broad claim?",
      "source_title": "The Weirdest People in the World?",
      "url": "https://superalignment.inc/library/weirdest-people-in-the-world/",
      "source": "https://doi.org/10.1017/S0140525X0999152X",
      "bottom_line": "A convenient sample can answer some questions and fail others. Showing that a behavior can occur may require only one population. Estimating how humans generally think or behave requires evidence across populations that could differ. Sample quality is therefore relative to claim scope, not a moral ranking of subjects or a blanket rejection of laboratory research.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "beyond-markets-and-states",
      "id": "doi:10.1257/aer.100.3.641",
      "work_id": "work:ostrom-beyond-markets-and-states",
      "title": "When can many governing centers work as one system?",
      "source_title": "Beyond Markets and States: Polycentric Governance of Complex Economic Systems",
      "url": "https://superalignment.inc/library/beyond-markets-and-states/",
      "source": "https://pubs.aeaweb.org/doi/10.1257/aer.100.3.641",
      "bottom_line": "Polycentricity is not a synonym for fragmentation or a claim that decentralization always wins. Its promise comes from several centers being able to learn, adapt, monitor, and constrain one another while remaining connected through rules and mechanisms that make their combined behavior a system. Whether that system is coherent and effective is an empirical question.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "communication-structure-of-epistemic-communities",
      "id": "doi:10.1086/525605",
      "work_id": "work:zollman-communication-structure-epistemic-communities",
      "title": "When can more scientific communication reduce reliability?",
      "source_title": "The Communication Structure of Epistemic Communities",
      "url": "https://superalignment.inc/library/communication-structure-of-epistemic-communities/",
      "source": "https://doi.org/10.1086/525605",
      "bottom_line": "More communication is not automatically more collective knowledge. A dense network can distribute a mistaken result so quickly that nobody continues the experiment needed to correct it. Sparse communication can preserve an epistemic firebreak, but it also delays useful evidence. The result is a conditional speed-reliability tradeoff, not a prescription to isolate researchers.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "psychological-safety-and-learning-behavior-in-work-teams",
      "id": "doi:10.2307/2666999",
      "work_id": "work:edmondson-psychological-safety-team-learning",
      "title": "Why do capable teams hide errors instead of learning from them?",
      "source_title": "Psychological Safety and Learning Behavior in Work Teams",
      "url": "https://superalignment.inc/library/psychological-safety-and-learning-behavior-in-work-teams/",
      "source": "https://doi.org/10.2307/2666999",
      "bottom_line": "A team can have competent members and still suppress the signals needed to improve if speaking up risks embarrassment, rejection, or punishment. Psychological safety lowers that interpersonal cost and can enable learning behavior. The paper does not equate safety with comfort, agreement, or low standards, and its cross-sectional one-company design cannot establish that increasing safety will cause performance gains in every team.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "risk-management-in-a-dynamic-society",
      "id": "doi:10.1016/s0925-7535(97)00052-0",
      "work_id": "work:rasmussen-risk-management-dynamic-society",
      "title": "How can ordinary adaptation move a system toward disaster?",
      "source_title": "Risk Management in a Dynamic Society: A Modelling Problem",
      "url": "https://superalignment.inc/library/risk-management-in-a-dynamic-society/",
      "source": "https://doi.org/10.1016/S0925-7535(97)00052-0",
      "bottom_line": "Safety can degrade through ordinary adaptation. Each local decision may make sense under its immediate information and incentives while their interaction moves the system toward a boundary nobody sees clearly. The remedy proposed is not tighter procedure alone, but control across levels with explicit safety constraints, usable feedback, and attention to how the whole system changes.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "multitask-principal-agent-analyses",
      "id": "doi:10.1093/jleo/7.special_issue.24",
      "work_id": "work:holmstrom-milgrom-multitask-principal-agent",
      "title": "When should a firm weaken a useful performance incentive?",
      "source_title": "Multitask Principal-Agent Analyses: Incentive Contracts, Asset Ownership, and Job Design",
      "url": "https://superalignment.inc/library/multitask-principal-agent-analyses/",
      "source": "https://doi.org/10.1093/jleo/7.special_issue.24",
      "bottom_line": "Do not set an incentive by inspecting the rewarded task alone. First map every task that competes for the agent's attention, how well each is measured, and which other controls shape the portfolio. The paper does not show that performance pay is generally harmful. It derives conditional results from a formal model whose strongest conclusions rely on linear contracts, substitutable attention, and other simplifying assumptions.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "viable-system-model-provenance-development-methodology-pathology",
      "id": "doi:10.1057/jors.1984.2",
      "work_id": "work:beer-viable-system-model",
      "title": "What makes an organization viable rather than merely alive?",
      "source_title": "The Viable System Model: Its Provenance, Development, Methodology and Pathology",
      "url": "https://superalignment.inc/library/viable-system-model-provenance-development-methodology-pathology/",
      "source": "https://doi.org/10.1057/jors.1984.2",
      "bottom_line": "The VSM's useful claim is that local execution, coordination, current control, future-facing intelligence, and identity are different regulatory jobs that must remain connected. Muting the future-facing function can collapse policy into short-term control and leave an organization reactive. That is a theoretical diagnostic, not independent evidence that five boxes guarantee survival.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "strategic-information-transmission",
      "id": "doi:10.2307/1913390",
      "work_id": "work:crawford-sobel-strategic-information-transmission",
      "title": "Why can informative communication still hide what matters?",
      "source_title": "Strategic Information Transmission",
      "url": "https://superalignment.inc/library/strategic-information-transmission/",
      "source": "https://doi.org/10.2307/1913390",
      "bottom_line": "Strategic communication often fails by coarsening rather than by obvious lying. A sender can reveal which broad region contains the state while suppressing the finer distinctions that would move the receiver against the sender's interest. The result is a model-conditional equilibrium claim, not a universal law that bias always destroys communication.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "assessing-impact-planned-social-change",
      "id": "doi:10.1016/0149-7189(79)90048-x",
      "work_id": "work:campbell-assessing-planned-social-change",
      "title": "Why do social metrics break when decisions depend on them?",
      "source_title": "Assessing the Impact of Planned Social Change",
      "url": "https://superalignment.inc/library/assessing-impact-planned-social-change/",
      "source": "https://doi.org/10.1016/0149-7189(79)90048-X",
      "bottom_line": "The paper does not say that every metric becomes useless when it matters. It says that consequential use creates pressure on both the record and the underlying process, so an evaluation needs independent criticism, contextual evidence, and designs that can expose alternative explanations rather than treating a score as a transparent window on performance.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "on-the-folly-of-rewarding-a-while-hoping-for-b",
      "id": "url:journals.aom.org/c934f11f2a",
      "work_id": "work:kerr-rewarding-a-hoping-b",
      "title": "Why do people optimize the reward instead of the stated goal?",
      "source_title": "On the Folly of Rewarding A, While Hoping for B",
      "url": "https://superalignment.inc/library/on-the-folly-of-rewarding-a-while-hoping-for-b/",
      "source": "https://journals.aom.org/doi/10.5465/255378",
      "bottom_line": "Before blaming motivation or culture, inspect the operative reward system from the recipient's point of view. A visible proxy can make A rational even while leaders announce B. Kerr does not claim that formal rewards determine every action, or that every apparent mismatch is a design error. Some cases reveal that leaders prefer A, while others reflect a real choice to prioritize equity or morality over efficiency.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "every-good-regulator-of-a-system-must-be-a-model-of-that-system",
      "id": "doi:10.1080/00207727008920220",
      "work_id": "work:good-regulator-theorem",
      "title": "What does the Good Regulator theorem actually prove?",
      "source_title": "Every good regulator of a system must be a model of that system",
      "url": "https://superalignment.inc/library/every-good-regulator-of-a-system-must-be-a-model-of-that-system/",
      "source": "https://doi.org/10.1080/00207727008920220",
      "bottom_line": "The proof establishes an outcome-equivalent deterministic state-action map under a fixed entropy objective. It does not establish that every effective regulator contains an explicit world model, that the mapping is one-to-one, or that the stabilized outcome is desirable.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "requisite-variety-complex-systems",
      "id": "url:ashby-requisite-variety-1958",
      "work_id": "work:ashby-requisite-variety-complex-systems",
      "title": "How much variety does a regulator need to control a system?",
      "source_title": "Requisite Variety and Its Implications for the Control of Complex Systems",
      "url": "https://superalignment.inc/library/requisite-variety-complex-systems/",
      "source": "https://pespmc1.vub.ac.be/books/AshbyReqVar.pdf",
      "bottom_line": "Requisite variety is an impossibility boundary: a regulator that cannot discriminate and answer enough relevant disturbance cases cannot guarantee a narrow outcome set. Crossing that boundary does not show that its goals are correct, its sensors are informative, its actions are effective, or its response policy pairs the right action with each disturbance.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    },
    {
      "slug": "federalist-no-51",
      "id": "url:founders-online-federalist-no-51",
      "work_id": "work:publius-federalist-no-51",
      "title": "How can a government be powerful without becoming unchecked?",
      "source_title": "The Federalist No. 51",
      "url": "https://superalignment.inc/library/federalist-no-51/",
      "source": "https://founders.archives.gov/documents/Hamilton/01-04-02-0199",
      "bottom_line": "The essay's durable design claim is conditional: checks work when institutions have separate bases of authority, usable constitutional means, and motives to defend their remit. Multiple branches or review bodies do not create restraint if one coalition controls their appointments, resources, information, and incentives.",
      "collection": "seminal-v1",
      "review_status": "prototype",
      "updated_at": "2026-08-17"
    }
  ]
}
