{"_fields":["title","authors","year","kind","venue","url_no_scheme","tier_code","slug_if_annotated","topics"],"_tier_codes":{"0":"seeded","1":"verified","2":"annotated"},"_tier_labels":{"0":"Listed","1":"Verified","2":"Explained"},"_license":"CC BY 4.0","_license_url":"https://creativecommons.org/licenses/by/4.0/","_license_scope":"Superalignment-authored compilation and editorial material only. Third-party source material retains its original rights.","_generated":"2026-09-08","_count":10618,"rows":[["Alignment Faking in Large Language Models","Ryan Greenblatt and 19 others","2024","paper","arXiv preprint arXiv:2412.14093","arxiv.org/abs/2412.14093",2,"alignment-faking-in-large-language-models","alignment-faking deception situational-awareness rlhf evals measurement agents"],["Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","Evan Hubinger and 38 others","2024","paper","arXiv preprint arXiv:2401.05566","arxiv.org/abs/2401.05566",2,"sleeper-agents-training-deceptive-llms-that-persist-through-safety-training","deception alignment-faking robustness red-teaming evals situational-awareness agents"],["AI Control: Improving Safety Despite Intentional Subversion","Ryan Greenblatt and 3 others","2024","paper","Proceedings of the 41st International Conference on Machine Learning, PMLR 235:16295-16336","proceedings.mlr.press/v235/greenblatt24a.html",2,"ai-control-improving-safety-despite-intentional-subversion","ai-control monitoring red-teaming deception sandbagging safety-cases capability-elicitation agents"],["Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision","Collin Burns and 11 others","2024","paper","Proceedings of the 41st International Conference on Machine Learning, PMLR 235:4971-5012","proceedings.mlr.press/v235/burns24b.html",2,"weak-to-strong-generalization","weak-to-strong scalable-oversight rlhf capability-elicitation measurement organizational-psychology"],["Towards Monosemanticity: Decomposing Language Models With Dictionary Learning","Trenton Bricken and 23 others","2023","report","Transformer Circuits Thread","transformer-circuits.pub/2023/monosemantic-features/index.html",2,"towards-monosemanticity","mechanistic-interpretability interpretability measurement training-data robustness"],["Scaling Laws for Reward Model Overoptimization","Leo Gao and 2 others","2023","paper","Proceedings of the 40th International Conference on Machine Learning, PMLR 202:10835-10866","proceedings.mlr.press/v202/gao23h.html",2,"scaling-laws-for-reward-model-overoptimization","rlhf reward-hacking goodharts-law scaling-laws measurement"],["Constitutional AI: Harmlessness from AI Feedback","Yuntao Bai and 50 others","2022","paper","arXiv preprint","arxiv.org/abs/2212.08073",2,"constitutional-ai-harmlessness-from-ai-feedback","constitutional-ai rlhf scalable-oversight cultural-values"],["Toy Models of Superposition","Nelson Elhage and 15 others","2022","blog","Transformer Circuits Thread","transformer-circuits.pub/2022/toy_model/index.html",2,"toy-models-of-superposition","mechanistic-interpretability interpretability theory measurement"],["Underspecification Presents Challenges for Credibility in Modern Machine Learning","Alexander D’Amour and 39 others","2022","paper","Journal of Machine Learning Research 23(226):1-61","www.jmlr.org/papers/v23/20-1335.html",2,"underspecification-presents-challenges-for-credibility-in-modern","benchmarks robustness training-data measurement"],["Goal Misgeneralization in Deep Reinforcement Learning","Lauro Langosco Di Langosco and 4 others","2022","paper","Proceedings of the 39th International Conference on Machine Learning, PMLR 162:12004-12019","proceedings.mlr.press/v162/langosco22a.html",2,"goal-misgeneralization-in-deep-reinforcement-learning","agents robustness measurement"],["The Impact of Network Connectivity on Collective Learning","Michael Crosscombe and Jonathan Lawry","2022","paper","Distributed Autonomous Robotic Systems: 15th International Symposium, DARS 2021, Springer Proceedings in Advanced Robotics 22, pages 82-94","arxiv.org/abs/2106.00655",2,"the-impact-of-network-connectivity-on-collective-learning","agents theory collective-learning network-science organizational-design social-epistemology"],["Training language models to follow instructions with human feedback","Long Ouyang and 19 others","2022","paper","Advances in Neural Information Processing Systems 35 (NeurIPS 2022)","proceedings.neurips.cc/paper_files/paper/2022/hash/b1efde53be364a73914f58805a001731-Abstract-Conference.html",2,"training-language-models-to-follow-instructions-with-human-feedback","rlhf scalable-oversight evals benchmarks measurement training-data robustness"],["Optimal Policies Tend To Seek Power","Alex Turner and 4 others","2021","paper","Advances in Neural Information Processing Systems 34, 23063-23074","proceedings.neurips.cc/paper/2021/hash/c26820b8a4c1b3c2aa868d6d57e14a79-Abstract.html",2,"optimal-policies-tend-to-seek-power","instrumental-convergence power-seeking agents theory"],["Eliciting latent knowledge: How to tell if your eyes deceive you","Paul Christiano and 2 others","2021","report","Alignment Research Center technical report","docs.google.com/document/d/1WwsnJQstPq91_Yh-Ch2XRL8H_EpsnjrC1dwZXR37PC8/edit",2,"eliciting-latent-knowledge-how-to-tell-if-your-eyes-deceive-you","eliciting-latent-knowledge scalable-oversight measurement interpretability"],["Algorithmic Monoculture and Social Welfare","Jon Kleinberg and Manish Raghavan","2021","paper","Proceedings of the National Academy of Sciences, 118(22), e2018340118","doi.org/10.1073/pnas.2018340118",2,"algorithmic-monoculture-and-social-welfare","theory governance game-theory mechanism-design algorithmic-monoculture institutional-design"],["Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective","Tom Everitt and 3 others","2021","paper","Synthese 198, Supplement 27, pages 6435-6467","link.springer.com/article/10.1007/s11229-021-03141-4",2,"reward-tampering-problems-and-solutions-in-reinforcement-learning-a","reward-hacking specification-gaming instrumental-convergence agents mechanism-design theory"],["Risks from Learned Optimization in Advanced Machine Learning Systems","Evan Hubinger and 4 others","2019","paper","arXiv preprint arXiv:1906.01820","arxiv.org/abs/1906.01820",2,"risks-from-learned-optimization-in-advanced-machine-learning-systems","agents deception situational-awareness robustness theory"],["Supervising strong learners by amplifying weak experts","Paul Christiano and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08575",2,"supervising-strong-learners-by-amplifying-weak-experts","scalable-oversight weak-to-strong measurement organizational-design automated-alignment-research"],["AI Safety via Debate","Geoffrey Irving and 2 others","2018","paper","arXiv preprint arXiv:1805.00899","arxiv.org/abs/1805.00899",2,"ai-safety-via-debate","scalable-oversight debate game-theory mechanism-design theory"],["Deep Reinforcement Learning from Human Preferences","Paul F. Christiano and 5 others","2017","paper","Advances in Neural Information Processing Systems 30 (NIPS 2017)","proceedings.neurips.cc/paper/2017/hash/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html",2,"deep-reinforcement-learning-from-human-preferences","rlhf scalable-oversight reward-hacking agents measurement"],["The Off-Switch Game","Dylan Hadfield-Menell and 3 others","2017","paper","Proceedings of the Twenty-Sixth International Joint Conference on Artificial Intelligence (IJCAI-17), pages 220-227","www.ijcai.org/proceedings/2017/32",2,"the-off-switch-game","game-theory mechanism-design agents instrumental-convergence ai-control theory"],["Concrete Problems in AI Safety","Dario Amodei and 5 others","2016","paper","arXiv preprint arXiv:1606.06565","arxiv.org/abs/1606.06565",2,"concrete-problems-in-ai-safety","reward-hacking specification-gaming scalable-oversight robustness monitoring safety-science theory"],["Cooperative Inverse Reinforcement Learning","Dylan Hadfield-Menell and 3 others","2016","paper","Advances in Neural Information Processing Systems 29 (NIPS 2016)","proceedings.neurips.cc/paper/2016/hash/c3395dd46c34fa7fd8d729d8cf88b7a8-Abstract.html",2,"cooperative-inverse-reinforcement-learning","game-theory mechanism-design agents theory scalable-oversight"],["Safely Interruptible Agents","Laurent Orseau and Stuart Armstrong","2016","paper","Proceedings of the Thirty-Second Conference on Uncertainty in Artificial Intelligence (UAI 2016), pages 557-566","www.auai.org/uai2016/proceedings.php",2,"safely-interruptible-agents","agents ai-control robustness instrumental-convergence control-theory theory"],["Engineering a Safer World: Systems Thinking Applied to Safety","Nancy G. Leveson","2012","book","MIT Press","doi.org/10.7551/mitpress/8179.001.0001",2,"engineering-a-safer-world","safety-science systems-theory control-theory organizational-design safety-cases assurance governance"],["The Weirdest People in the World?","Joseph Henrich and 2 others","2010","paper","Behavioral and Brain Sciences, 33(2-3), 61-83","doi.org/10.1017/S0140525X0999152X",2,"weirdest-people-in-the-world","cultural-values social-epistemology measurement pluralistic-alignment theory"],["Beyond Markets and States: Polycentric Governance of Complex Economic Systems","Elinor Ostrom","2010","paper","American Economic Review, 100(3), 641-672","pubs.aeaweb.org/doi/10.1257/aer.100.3.641",2,"beyond-markets-and-states","institutional-design governance public-administration game-theory mechanism-design social-epistemology organizational-design"],["The Communication Structure of Epistemic Communities","Kevin J. S. Zollman","2007","paper","Philosophy of Science, 74(5), 574-587","doi.org/10.1086/525605",2,"communication-structure-of-epistemic-communities","social-epistemology network-science collective-learning organizational-design theory"],["Psychological Safety and Learning Behavior in Work Teams","Amy Edmondson","1999","paper","Administrative Science Quarterly, 44(2), 350-383","doi.org/10.2307/2666999",2,"psychological-safety-and-learning-behavior-in-work-teams","organizational-psychology collective-learning social-epistemology organizational-design safety-science"],["Risk Management in a Dynamic Society: A Modelling Problem","Jens Rasmussen","1997","paper","Safety Science, 27(2-3), 183-213","doi.org/10.1016/S0925-7535(97)00052-0",2,"risk-management-in-a-dynamic-society","safety-science systems-theory control-theory organizational-design regulatory-design public-administration governance"],["Multitask Principal-Agent Analyses: Incentive Contracts, Asset Ownership, and Job Design","Bengt Holmstrom and Paul Milgrom","1991","paper","Journal of Law, Economics, & Organization, 7(Special Issue), 24-52","doi.org/10.1093/jleo/7.special_issue.24",2,"multitask-principal-agent-analyses","measurement game-theory mechanism-design organizational-design institutional-design theory"],["The Viable System Model: Its Provenance, Development, Methodology and Pathology","Stafford Beer","1984","paper","Journal of the Operational Research Society, 35(1), 7-25","doi.org/10.1057/jors.1984.2",2,"viable-system-model-provenance-development-methodology-pathology","cybernetics organizational-design systems-theory control-theory institutional-design governance"],["Strategic Information Transmission","Vincent P. Crawford and Joel Sobel","1982","paper","Econometrica, 50(6), 1431-1451","doi.org/10.2307/1913390",2,"strategic-information-transmission","game-theory mechanism-design theory social-epistemology organizational-design"],["Assessing the Impact of Planned Social Change","Donald T. Campbell","1979","paper","Evaluation and Program Planning, 2(1), 67-90","doi.org/10.1016/0149-7189(79)90048-X",2,"assessing-impact-planned-social-change","measurement goodharts-law evals institutional-design public-administration organizational-psychology"],["On the Folly of Rewarding A, While Hoping for B","Steven Kerr","1975","paper","Academy of Management Journal, 18(4), 769-783","journals.aom.org/doi/10.5465/255378",2,"on-the-folly-of-rewarding-a-while-hoping-for-b","organizational-psychology measurement goodharts-law mechanism-design institutional-design organizational-design"],["Every good regulator of a system must be a model of that system","Roger C. Conant and W. Ross Ashby","1970","paper","International Journal of Systems Science, 1(2), 89-97","doi.org/10.1080/00207727008920220",2,"every-good-regulator-of-a-system-must-be-a-model-of-that-system","cybernetics systems-theory control-theory regulatory-design constitutional-design"],["Requisite Variety and Its Implications for the Control of Complex Systems","W. Ross Ashby","1958","paper","Cybernetica, 1(2), 83-99","pespmc1.vub.ac.be/books/AshbyReqVar.pdf",2,"requisite-variety-complex-systems","cybernetics systems-theory control-theory regulatory-design organizational-design theory"],["The Federalist No. 51","Publius","1788","constitution","Independent Journal","founders.archives.gov/documents/Hamilton/01-04-02-0199",2,"federalist-no-51","constitutional-design institutional-design governance public-administration mechanism-design game-theory"],["Does DiffusionGemma do latent reasoning?","Jan Bauer and Neel Nanda","2026","blog","AI Alignment Forum","www.alignmentforum.org/posts/QBuJ3suRZxrrxSTtv/does-diffusiongemma-do-latent-reasoning",1,"","interpretability chain-of-thought-faithfulness"],["Behavioral Reprogramming of Open-Weights Models: Cognitive Plasticity and Alignment Bounds","Lucia Malíčková","2026","paper","arXiv preprint","arxiv.org/abs/2608.13069",1,"","robustness"],["Practice Makes Unsafe: Skill Misevolution in Self-Improving LLM Agents","Xutao Mao and 3 others","2026","paper","arXiv preprint","arxiv.org/abs/2608.12851",1,"","agents"],["Rules or Character? Scaling Laws for AI Safety Design","Satoshi Takahashi and 3 others","2026","paper","arXiv preprint","arxiv.org/abs/2608.13345",1,"","scaling-laws"],["AI swarms are starting to pose indirect takeover risk","oakhu and Alex Mallen","2026","blog","AI Alignment Forum","www.alignmentforum.org/posts/8oFYZdXkTaNGRtcn8/ai-swarms-are-starting-to-pose-indirect-takeover-risk",1,"","agents"],["Introducing the Conceptual Reasoning Index","Emery Cooper and 5 others","2026","benchmark","Alignment Science Blog","alignment.anthropic.com/2026/conceptual-reasoning-index/",1,"","scalable-oversight evals benchmarks robustness"],["GPT-Red: Automated Red Teaming via Self-Play at Scale","Eric Wallace and 17 others","2026","paper","arXiv preprint","arxiv.org/abs/2607.26115",1,"","red-teaming robustness"],["How independent researchers could investigate AI propensities after misalignment incidents","METR","2026","blog","METR Blog","metr.org/blog/2026-07-28-investigating-ai-propensities-after-incidents/",1,"","agents"],["Agentic Misalignment in Summer 2026","Aengus Lynch (Theorem; work done as part of the Anthropic Fellows program) and 4 others","2026","blog","Alignment Science Blog","alignment.anthropic.com/2026/agentic-misalignment-summer-2026/",1,"","agents monitoring"],["Modular Pretraining Enables Access Control","Ethan Roland and 10 others","2026","blog","Alignment Science Blog","alignment.anthropic.com/2026/modular-pretraining/",1,"",""],["Separating signal from noise in coding evaluations","OpenAI","2026","blog","OpenAI website","openai.com/index/separating-signal-from-noise-coding-evaluations/",1,"","evals benchmarks"],["Verbalizable Representations Form a Global Workspace in Language Models","Wes Gurnee * and 15 others","2026","paper","Transformer Circuits Thread","transformer-circuits.pub/2026/workspace/index.html",1,"","interpretability mechanistic-interpretability"],["Summary of METR's predeployment evaluation of GPT-5.6 Sol","METR","2026","blog","METR Blog","metr.org/blog/2026-06-26-gpt-5-6-sol/",1,"","evals"],["\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","Alan Cooney and 2 others","2026","paper","arXiv preprint","arxiv.org/abs/2606.12618",1,"","evals deception monitoring chain-of-thought-faithfulness"],["Diffuse AI Control on Fuzzy Tasks","Mikhail Terekhov and 3 others","2026","paper","arXiv preprint","arxiv.org/abs/2606.08892",1,"","scalable-oversight ai-control sandbagging automated-alignment-research"],["Quantifying the Salience of Geo-Cultural Values for Pluralistic Safety Alignment","Arkadiy Saakyan and 2 others","2026","paper","arXiv preprint","arxiv.org/abs/2606.00369",1,"","evals"],["Gram: Assessing sabotage propensities via automated alignment auditing","David Lindner and 2 others","2026","paper","arXiv","deepmind.google/research/publications/252981/",1,"","scalable-oversight automated-alignment-research"],["Realistic honeypot evaluations for scheming propensity","Victoria Krakovna and 4 others","2026","paper","arXiv","deepmind.google/research/publications/253391/",1,"","evals deception"],["SLEIGHT-Bench: A Benchmark of Evasion Attacks Against Agent Monitors","Elle Najt and 4 others","2026","paper","arXiv preprint","arxiv.org/abs/2605.16626",1,"","ai-control benchmarks red-teaming agents monitoring"],["Automated alignment is harder than you think","Aleksandr Bowkis and 3 others","2026","paper","arXiv preprint","arxiv.org/abs/2605.06390",1,"","scalable-oversight evals automated-alignment-research assurance"],["Natural Language Autoencoders Produce Unsupervised Explanations of LLM Activations","Kit Fraser-Taliente and 19 others","2026","paper","Transformer Circuits Thread","transformer-circuits.pub/2026/nla/index.html",1,"","interpretability mechanistic-interpretability"],["Model Spec Midtraining: Improving How Alignment Training Generalizes","Chloe Li and 4 others","2026","paper","arXiv preprint","arxiv.org/abs/2605.02087",1,"","agents robustness"],["ProEval: Proactive Failure Discovery and Efficient Performance Estimation for Generative AI Evaluation","Yizheng Huang and 3 others","2026","paper","arXiv preprint","arxiv.org/abs/2604.23099",1,"","evals benchmarks"],["BashArena: A Control Setting for Highly Privileged AI Agents","Adam Kaufman and 4 others","2025","paper","arXiv preprint","arxiv.org/abs/2512.15688",1,"","ai-control evals agents"],["Imitation Learning is Probably Existentially Safe","Michael K. Cohen and Marcus Hutter","2025","paper","AI Magazine","deepmind.google/research/publications/42697/",1,"",""],["Ctrl-Z: Controlling AI Agents via Resampling","Aryan Bhatt and 7 others","2025","paper","arXiv preprint","arxiv.org/abs/2504.10374",1,"","agents"],["80k's Podcast","","","report","80000hours.org","80000hours.org/podcast/",0,"",""],["AI Alignment Forum Sequences","","","report","drive.google.com","drive.google.com/open?id=1qnBEfb-TvnvVlJ4cx81cAXg_XcQbG02rcPb1r6T28Kk",0,"",""],["Aisafety.video","","","report","aisafety.video","aisafety.video/",0,"",""],["Algorithmic learning in a random world","Vladimir Vovk and 2 others","","report","alrw.net","www.alrw.net/",0,"",""],["An Introduction to Game Theory, Chapters 1-7,14,15","Osborne","","report","global.oup.com","global.oup.com/ushe/product/an-introduction-to-game-theory-9780195128956?cc=gb&lang=en&",0,"",""],["Antifragile","","","report","youtube.com","www.youtube.com/watch?v=-MMLea-_ifw",0,"",""],["Artificial Intelligence Safety and Security","","","report","routledge.com","www.routledge.com/Artificial-Intelligence-Safety-and-Security/Yampolskiy/p/book/9780815369820",0,"",""],["Bayesian Data Analysis","Gelman and Rubin","","report","stat.columbia.edu","www.stat.columbia.edu/~gelman/book/",0,"",""],["Beyond Normal Accidents and High Reliability Organizations: The Need for an Alternative Approach to Safety in Complex Systems","","","report","sunnyday.mit.edu","sunnyday.mit.edu/papers/hro.pdf",0,"",""],["Causality: models, reasoning, and inference","Judea Pearl","","report","amazon.com","www.amazon.com/Causality-Reasoning-Inference-Judea-Pearl/dp/052189560X",0,"",""],["CEA Artificial Intelligence","","","report","youtube.com","www.youtube.com/watch?v=dbMp4pFVwnU&list=PLwp9xeoX5p8P0TuCSdTExdeCKr97DqXWZ",0,"",""],["CEA's Existential Risk and the Far Future playlist","","","report","youtube.com","www.youtube.com/watch?v=l6yAylvzEXo&list=PLwp9xeoX5p8NjWAeGnbe5tQwoXm3oMY3H",0,"",""],["Center for AI Safety","","","report","safe.ai","safe.ai",0,"",""],["Circuits Updates - July 2023","","","blog","transformer-circuits.pub","transformer-circuits.pub/2023/july-update/index.html",0,"","mechanistic-interpretability"],["Circuits Updates — May 2023","","","blog","transformer-circuits.pub","transformer-circuits.pub/2023/may-update/index.html",0,"","mechanistic-interpretability"],["CLR's YouTube channel","","","report","youtube.com","www.youtube.com/channel/UCNPqscTt41xxJ-8RCN_3vFA",0,"",""],["Computability and Logic, Chapters 1-4, 8-20, 23, 25, and 27","Boolos and Burgess","","report","cambridge.org","www.cambridge.org/us/academic/subjects/philosophy/logic/computability-and-logic-5th-edition",0,"",""],["Corrigibility","Nate Soares and 3 others","","report","aaai.org","aaai.org/ocs/index.php/WS/AAAIW15/paper/view/10124/10136",0,"",""],["CSER's YouTube channel","","","report","youtube.com","www.youtube.com/c/CSERCambridge/",0,"",""],["Dan Hendrycks","","","report","danhendrycks.com","danhendrycks.com/",0,"",""],["Deep Learning, Chapters 1-5","Goodfellow and 2 others","","report","deeplearningbook.org","www.deeplearningbook.org/contents/part_basics.html",0,"",""],["Deep Learning, Chapters 6-12","Goodfellow and 2 others","","report","deeplearningbook.org","www.deeplearningbook.org/contents/part_practical.html",0,"",""],["Deep Reinforcement Learning","Schulman","","report","youtube.com","www.youtube.com/watch?v=9dXiAecyJrY",0,"",""],["DeepMind's YouTube channel","","","report","youtube.com","www.youtube.com/channel/UCP7jMXSY2xbc3KCAE0MHQ-A",0,"",""],["Defining Human Values for Value Learners","Kaj Sotala","","report","aaai.org","www.aaai.org/ocs/index.php/WS/AAAIW16/paper/download/12633/12353",0,"",""],["Distributed Representations: Composition & Superposition","","","blog","transformer-circuits.pub","transformer-circuits.pub/2023/superposition-composition/index.html",0,"","mechanistic-interpretability"],["Do Androids Dream of Electric Sheep?","Philip K. Dick","","report","amazon.com","www.amazon.com/Androids-Dream-Electric-Sheep-inspiration/dp/0345404475",0,"",""],["Dropout: A Simple Way to Prevent Neural Networks from Overfitting","","","report","jmlr.org","jmlr.org/papers/v15/srivastava14a.html",0,"",""],["Emergence (Basic)","","","report","youtube.com","www.youtube.com/watch?v=16W7c0mb-rE",0,"",""],["Emergence (Intermediate)","","","report","youtube.com","www.youtube.com/watch?v=QItTWZc7hKs",0,"",""],["Empirical examples of power-law CDFs","","","report","youtu.be","youtu.be/KKYhPPf4FxA?t=345",0,"",""],["Ethics Background (Introduction through “Absolute Rights or Prima Facie Duties”)","","","report","youtube.com","www.youtube.com/playlist?list=PLKtXFotbf7fOg7zbQ3565EnpzzKlYaVVI",0,"",""],["Everything else Chris Olah has ever written","","","report","colah.github.io","colah.github.io/",0,"",""],["Flatland: A Romance of Many Dimensions","Edwin A. Abbott","","report","gutenberg.org","www.gutenberg.org/ebooks/201",0,"",""],["FLI's AI Safety Research Landscape","","","report","futureoflife.org","futureoflife.org/landscape/",0,"",""],["Formalizing Convergent Instrumental Goals","Tsvi Benson-Tilsen and Nate Soares","","report","aaai.org","www.aaai.org/ocs/index.php/WS/AAAIW16/paper/view/12634/12347",0,"","instrumental-convergence"],["Game Theory I (Coursera)","Jackson and 2 others","","report","coursera.org","www.coursera.org/learn/game-theory-1",0,"",""],["Game Theory II (Coursera)","Jackson and 2 others","","report","coursera.org","www.coursera.org/learn/game-theory-2",0,"",""],["Game Theory: Analysis of Conflict","Roger Myerson","","report","amazon.com","www.amazon.com/Game-Theory-Analysis-Roger-Myerson/dp/0674341163/",0,"",""],["Geometry for Ocelots","Exurb1a","","report","amazon.com","www.amazon.com/gp/product/B0969DPK7H/ref=x_gr_bb_amazon?ie=UTF8&tag=x_gr_bb_amazon-20&linkCode=as2&camp=1789&creative=9325&creativeASIN=B0969DPK7H&SubscriptionId=1MGPYB6YW3HWK55XCGG2",0,"",""],["Governance of AI program","","","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/governance-ai-program/",0,"","governance"],["Handbook of Model Checking (to appear soon)","Clarke and 2 others","","report","springer.com","www.springer.com/us/book/9783319105741",0,"",""],["Harry Potter and the Methods of Rationality (#1 of 6)","Eliezer Shlomo Yudkowsky","","report","hpmor.com","www.hpmor.com/",0,"",""],["How Bad is Selfish Voting?","Simina Branzei and 3 others","","report","cis.upenn.edu","www.cis.upenn.edu/~jamiemor/papers/aaai-13.pdf",0,"",""],["Human Compatible","Stuart Russell","","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/hc.html",0,"",""],["Information Theory, Inference, and Learning Algorithms Parts I-III","McKay","","report","inference.phy.cam.ac.uk","www.inference.phy.cam.ac.uk/itila/",0,"",""],["Interpretability Dreams","","","blog","transformer-circuits.pub","transformer-circuits.pub/2023/interpretability-dreams/index.html",0,"","interpretability"],["Introduction to AI","Pieter Abbeel and Dan Klein","","report","ai.berkeley.edu","ai.berkeley.edu/lecture_videos.html",0,"",""],["Introduction to Artificial Intelligence","Norvig and Thrun","","report","udacity.com","www.udacity.com/course/intro-to-artificial-intelligence--cs271",0,"",""],["Introduction to Automata Theory, Languages, and Computation, Chapters 1-10","Ullman and Hopcroft","","report","infolab.stanford.edu","infolab.stanford.edu/~ullman/ialc.html",0,"",""],["Introduction to Probability","Richard S. Sutton and Andrew G. Barto","","report","edx.org","www.edx.org/course/introduction-probability-science-mitx-6-041x-1",0,"",""],["Learning the Preferences of Bounded Agents","Owain Evans and 2 others","","report","web.mit.edu","web.mit.edu/owain/www/nips-workshop-2015-website.pdf",0,"","agents"],["Log-normal distributions (with comparisons to power laws)","","","report","youtu.be","youtu.be/KiTAyRZORtQ?t=124",0,"",""],["Logic Beach","Exurb1a","","report","amazon.com","www.amazon.com/gp/product/B077SDRMHR",0,"",""],["Machine Learning","Nando de Freitas","","report","youtube.com","www.youtube.com/watch?v=w2OtwL5T1ow&list=PLE6Wd9FR--EdyJ5lbFl8UuGjecvVw66F6",0,"",""],["Machine Learning (lecture notes)","Andrew Ng","","report","cs229.stanford.edu","cs229.stanford.edu/materials.html",0,"",""],["Machine Learning (online course)","Andrew Ng","","report","coursera.org","www.coursera.org/learn/machine-learning",0,"",""],["Mathematical Logic : A course with exercises — Part I","Cori and Lascar","","report","amazon.com","www.amazon.com/Mathematical-Logic-exercises-Propositional-Completeness/dp/0198500483",0,"",""],["Mathematical Logic : A course with exercises — Part II, Chapters 5 and 6","Cori and Lascar","","report","amazon.com","www.amazon.com/Recursion-Theory-Godels-Theorems-Mathematical/dp/0198500505/",0,"",""],["Maximizing a quantity while ignoring effect through some channel","Jessica Taylor and Chris Olah","","report","agentfoundations.org","agentfoundations.org/item?id=735",0,"",""],["Mechanistic Interpretability, Variables, and the Importance of Interpretable Bases","","","blog","transformer-circuits.pub","transformer-circuits.pub/2022/mech-interp-essay/index.html",0,"","interpretability mechanistic-interpretability"],["MIRI's YouTube channel","","","report","youtube.com","www.youtube.com/c/IntelligenceOrg/",0,"",""],["Moral Machines: Teaching robots right from wrong","Wendell Wallach","","report","amazon.com","www.amazon.com/Moral-Machines-Teaching-Robots-Right/dp/0195374045/r",0,"",""],["Motivated Value Selection for Artificial Agents","Stuart Armstrong","","report","aaai.org","www.aaai.org/ocs/index.php/WS/AAAIW15/paper/viewFile/10183/10126",0,"","agents"],["Multiplicative processes produce log normals","","","report","youtube.com","www.youtube.com/watch?v=yA1pkyanbzw",0,"",""],["Multiplicative processes produce power laws","","","report","youtu.be","youtu.be/B43yhWxdi5I?t=69",0,"",""],["Neural Networks","Hugo Larochelle","","report","youtube.com","www.youtube.com/playlist?list=PL6Xpj9I5qXYEcOhn7TqghAJ6NAPrNmUBH",0,"",""],["Nonlinear Causality","","","report","youtube.com","www.youtube.com/watch?v=76JRJ90s548",0,"",""],["Not a paper: The codebase of EasyTransformer, a transformer mechanistic interpretability I’m writing - I think it’s worth reading for a fairly clean and conceptual-focused implementation of a transformer, specifically reading EasyTransformer.forward and components.py (a file for the various layers) (the actual codebase is pretty long!)","","","report","github.com","github.com/neelnanda-io/Easy-Transformer",0,"","interpretability mechanistic-interpretability"],["Numerous power laws for cities","","","report","youtube.com","www.youtube.com/watch?v=DsL7jEQXh8I",0,"",""],["Of Ants and Dinosaurs","Cixin Liu","","report","amazon.com","www.amazon.com/gp/product/B00838GX52",0,"",""],["On Explainability in Machine Learning","David Bianco","","report","mlsecproject.org","www.mlsecproject.org/blog/on-explainability-in-machine-learning",0,"",""],["OpenAI's YouTube channel","","","report","youtube.com","www.youtube.com/c/OpenAI/",0,"",""],["Pattern Recognition and Machine Learning","Bishop","","report","springer.com","www.springer.com/us/book/9780387310732",0,"",""],["Probability: Theory and Examples, Chapters 1-6","Durrett","","report","services.math.duke.edu","services.math.duke.edu/~rtd/PTE/pte.html",0,"",""],["Quantilizers Limited Optimization","Jessica Taylor","","report","aaai.org","www.aaai.org/ocs/index.php/WS/AAAIW16/paper/download/12613/12354",0,"",""],["Reinforcement Learning","Silver","","report","youtube.com","www.youtube.com/watch?v=2pWv7GOvuf0&list=PL5X3mDkKaJrL42i_jhE4N-p6E2Ol62Ofa",0,"",""],["Review of properties of power laws","","","report","youtube.com","www.youtube.com/watch?v=9Hc227Qy91k",0,"",""],["Robert Miles discuss AI on Computerphile","","","report","youtube.com","www.youtube.com/watch?v=tlS5Y2vm02c&list=PLzH6n4zXuckquVnQ0KlMDxyT5YE-sA8Ps",0,"",""],["Robert Miles's own YouTube channel","","","report","youtube.com","www.youtube.com/channel/UCLB7AzTwc6VFZrBsO2ucBMg",0,"",""],["Rust Circuit Library","Contributors: Adrià Garriga-Alonso and 9 others","","report","github.com","github.com/redwoodresearch/rust_circuit_public",0,"","mechanistic-interpretability"],["SlateStarCodex Meetups","","","report","youtube.com","www.youtube.com/watch?v=Wn2vgQGNI_c&list=PLFDYxsqlH6uhSghWfsuEAiKDfZNVZhUOX",0,"",""],["Superforecasting – Philip Tetlock","","","report","youtu.be","youtu.be/pedNak4S9IE?t=440",0,"","forecasting"],["Superintelligence","Nick Bostrom","","report","amazon.com","www.amazon.com/gp/product/0199678111",0,"",""],["Technical AI Safety Podcast","","","report","technical-ai-safety.libsyn.com","technical-ai-safety.libsyn.com/",0,"",""],["The Age of Em","Robin Hanson","","report","smile.amazon.co.uk","smile.amazon.co.uk/Age-Em-Work-Robots-Earth/dp/0198754620?sa-no-redirect=1",0,"",""],["The Alignment Problem","Brain Christian","","report","brianchristian.org","brianchristian.org/the-alignment-problem/",0,"",""],["The Black Swan","","","report","youtube.com","www.youtube.com/watch?v=caPy0OZmXKs",0,"",""],["The Bridge to Lucy Dunne","Exurb1a","","report","amazon.com","www.amazon.com/gp/product/B01F7IQEHC/",0,"",""],["The Dark Forest (#2 of Three Body Problem)","Cixin Liu","","report","amazon.com","www.amazon.com/The-Dark-Forest-audiobook/dp/B010R28SZ4",0,"",""],["The Inside View","","","report","youtube.com","www.youtube.com/channel/UCb9F9_uV24PGj6x63PhXEVw",0,"",""],["The Nonlinear Library","","","report","podcasts.google.com","podcasts.google.com/feed/aHR0cHM6Ly9hdWRpby5iZXlvbmR3b3Jkcy5pby9mLzkxOTUvMzAyNjUvcmVhZF84NjE3ZDNhZWU1M2YzYWI4NDRhMzA5ZDM3ODk1YzE0Mw",0,"",""],["The Righteous Mind: Why Good People Are Divided by Politics and Religion","Jonathan Haidt","","report","amazon.com","www.amazon.com/Righteous-Mind-Divided-Politics-Religion/dp/0307455777",0,"","robustness"],["The Singularity is Near","Ray Kurzweil","","report","smile.amazon.com","smile.amazon.com/The-Singularity-Is-Near-audiobook/dp/B07XPFT63D/ref=sr_1_1?crid=1CDY6DV8VCXWR&keywords=the+singularity+is+near&qid=1645680963&s=audible&sprefix=the+singularity+is+nea%2Caudible%2C118&sr=1-1\\n",0,"",""],["The Singularity: A Philosophical Analysis","David Chalmers","","report","ingentaconnect.com","www.ingentaconnect.com/content/imp/jcs/2010/00000017/f0020009/art00001",0,"",""],["The Structure of Normative Ethics","","","report","cpb-us-west-2-juc1ugur1qwqqqo4.stackpathdns.com","cpb-us-west-2-juc1ugur1qwqqqo4.stackpathdns.com/campuspress.yale.edu/dist/7/724/files/2016/01/The-Structure-of-Normative-Ethics-2gw2akt.pdf",0,"",""],["The Unilateralist’s Curse: The Case for a Principle of Conformity","Nick Bostrom and 2 others","","report","s3.amazonaws.com","s3.amazonaws.com/academia.edu.documents/43194856/The_Unilateralist_s_Curse_and_the_Case_for_a_Principle_of_Conformity.pdf?AWSAccessKeyId=AKIAJ56TQJRTWSMTNPEA&Expires=1479107896&Signature=1yHmcm94SI8hagp34v55Oc48Q08%3D&response-content-disposition=inline%3B%20filename%3DThe_Unilateralists_Curse_and_the_Case_fo.pdf",0,"",""],["Theory and applications of Robust Optimization","Dimitris Bertsimas","","paper","arXiv preprint","arxiv.org/abs/1010.5445",0,"",""],["Transformer Circuit Exercises","","","blog","transformer-circuits.pub","transformer-circuits.pub/2021/exercises/index.html",0,"","mechanistic-interpretability"],["Transformer Circuit Videos","","","blog","transformer-circuits.pub","transformer-circuits.pub/2021/videos/index.html",0,"","mechanistic-interpretability"],["Universal Artificial Intelligence, Chapters 2-5","Hutter","","report","hutter1.net","www.hutter1.net/ai/uaibook.htm",0,"",""],["We Are Legion (We Are Bob)","Dennis E. Taylor","","report","amazon.com","www.amazon.com/gp/product/B01LWAESYQ",0,"",""],["What happens when our computers get smarter than we are?","","","report","youtube.com","www.youtube.com/watch?v=MnT1xgZgkpk",0,"",""],["What is a Complex System?","","","report","youtube.com","www.youtube.com/watch?v=vp8v2Udd_PM",0,"",""],["Workshop On Safety And Control For Artificial Intelligence","","","report","cmu.edu","www.cmu.edu/safartint/watch.html",0,"",""],["Experiences and learnings from both sides of the AI safety job market","Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WqYSmjSsE3hi8Lgot/experiences-and-learnings-from-both-sides-of-the-ai-safety",0,"",""],["Incidental polysemanticity","Victor Lecomte and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sEyWufriufTnBKnTG/incidental-polysemanticity",0,"","interpretability"],["LLMs May Find It Hard to FOOM","RogerDearnaley","2023","blog","LessWrong","www.lesswrong.com/posts/qGTxGGNxcciY2nrHv/llms-may-find-it-hard-to-foom",0,"","forecasting"],["New report: \"Scheming AIs: Will AIs fake alignment during training in order to get power?\"","Joe Carlsmith","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yFofRxg7RRQYCcwFA/new-report-scheming-ais-will-ais-fake-alignment-during",0,"",""],["Betting on what is un-falsifiable and un-verifiable","Abhimanyu Pallavi Sudhir","2023","blog","LessWrong","www.lesswrong.com/posts/id84oe3LxdzoqinKA/betting-on-what-is-un-falsifiable-and-un-verifiable",0,"","eliciting-latent-knowledge"],["Is Interpretability All We Need?","RogerDearnaley","2023","blog","LessWrong","www.lesswrong.com/posts/tQt9pF9pbAuaiHxE6/is-interpretability-all-we-need",0,"","interpretability"],["Is there Work on Embedded Agency in Cellular Automata Toy Models?","Johannes C. Mayer","2023","blog","LessWrong","www.lesswrong.com/posts/ryhzvdgHEH77QTzcb/is-there-work-on-embedded-agency-in-cellular-automata-toy",0,"","theory"],["Would this be Progress in Solving Embedded Agency?","Johannes C. Mayer","2023","blog","LessWrong","www.lesswrong.com/posts/Ls2i4fgbEy9XarxzW/would-this-be-progress-in-solving-embedded-agency",0,"","theory"],["AISC Project: Benchmarks for Stable Reflectivity","jacquesthibs","2023","blog","LessWrong","www.lesswrong.com/posts/RHojGPWLgdFLk3PAt/aisc-project-benchmarks-for-stable-reflectivity",0,"","benchmarks"],["AISC Project: Modelling Trajectories of Language Models","NickyP","2023","blog","LessWrong","www.lesswrong.com/posts/JnmouffwMTYmRnoxT/aisc-project-modelling-trajectories-of-language-models",0,"","interpretability"],["Optionality approach to ethics","Ryo","2023","blog","LessWrong","www.lesswrong.com/posts/Ncv5b2sjLtyi9oKGz/optionality-approach-to-ethics",0,"",""],["Out of the Box","jesseduffield","2023","blog","LessWrong","www.lesswrong.com/posts/FFA6b4NoxaWYcbcZH/out-of-the-box",0,"",""],["The Science Algorithm AISC Project","Johannes C. Mayer","2023","blog","LessWrong","www.lesswrong.com/posts/KHjQzxRnDCjM7xsFk/the-science-algorithm-aisc-project",0,"",""],["Theories of Change for AI Auditing","Lee Sharkey and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LwJwDNFhjurAKFiJm/theories-of-change-for-ai-auditing",0,"","evals governance"],["Why small phenomenons are relevant to morality ​","Ryo","2023","blog","LessWrong","www.lesswrong.com/posts/4JvnwryM8rGiPmWBy/why-small-phenomenons-are-relevant-to-morality-1",0,"",""],["AISC project: SatisfIA – AI that satisfies without overdoing it","Jobst Heitzig","2023","blog","LessWrong","www.lesswrong.com/posts/ip2Cqas89TxsWok3S/aisc-project-satisfia-ai-that-satisfies-without-overdoing-it",0,"","agents"],["Control Symmetry: why we might want to start investigating asymmetric alignment interventions","domenicrosati","2023","blog","LessWrong","www.lesswrong.com/posts/JbmDxh5WLjX8AiTQv/control-symmetry-why-we-might-want-to-start-investigating",0,"",""],["Game Theory without Argmax [Part 1]","Cleo Nardo","2023","blog","LessWrong","www.lesswrong.com/posts/3ahqzpKvtqkom63cx/game-theory-without-argmax-part-1",0,"","agents theory"],["Game Theory without Argmax [Part 2]","Cleo Nardo","2023","blog","LessWrong","www.lesswrong.com/posts/BtZSNfAcBGQftAwxq/game-theory-without-argmax-part-2",0,"","agents theory"],["Open Phil releases RFPs on LLM Benchmarks and Forecasting","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ccNggNeBgMZFy3FRr/open-phil-releases-rfps-on-llm-benchmarks-and-forecasting",0,"","benchmarks forecasting"],["The Existential Risk of Speciesist Bias in AI","Sam Tucker","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gpNZbrSjHMHYqhvHn/the-existential-risk-of-speciesist-bias-in-ai",0,"",""],["The Top AI Safety Bets for 2023: GiveWiki’s Latest Recommendations","Dawn Drescher","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bEe4nRbShq8sWEE7n/the-top-ai-safety-bets-for-2023-givewiki-s-latest",0,"","forecasting"],["AI Timelines","habryka and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/K2D45BNxnZjdpSX2j/ai-timelines",0,"","forecasting"],["Artefacts generated by mode collapse in GPT-4 Turbo serve as adversarial attacks.","Sohaib Imran","2023","blog","LessWrong","www.lesswrong.com/posts/nxhXTfsAf2LTg4xvt/artefacts-generated-by-mode-collapse-in-gpt-4-turbo-serve-as",0,"","rlhf robustness"],["EA Poland is facing an existential risk","EA Poland","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wcXrW2cyi2zkJxDmo/ea-poland-is-facing-an-existential-risk",0,"",""],["GPT-2030 and Catastrophic Drives: Four Vignettes","jsteinhardt","2023","blog","LessWrong","www.lesswrong.com/posts/acPYHjC9euGZRzaj6/gpt-2030-and-catastrophic-drives-four-vignettes",0,"","forecasting"],["Munk Debate on AI: a few observations and opinions","Yarrow Bouchard","2023","blog","LessWrong","www.lesswrong.com/posts/Lx4BfG4kjNqxzfbt9/munk-debate-on-ai-a-few-observations-and-opinions",0,"",""],["Update on the UK AI Summit and the UK's Plans","Elliot_Mckernon","2023","blog","LessWrong","www.lesswrong.com/posts/gWwMzAgDsskcb2deA/update-on-the-uk-ai-summit-and-the-uk-s-plans",0,"","governance"],["We have promising alignment plans with low taxes","Seth Herd","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/xqqhwbH2mq6i4iLmK/we-have-promising-alignment-plans-with-low-taxes",0,"","chain-of-thought-faithfulness"],["ACI#6: A Non-Dualistic ACI Model","Akira Pyinya","2023","blog","LessWrong","www.lesswrong.com/posts/FRd6nNj3M33w2CSX5/aci-6-a-non-dualistic-aci-model",0,"","theory"],["Into AI Safety Episodes 1 & 2","jacobhaimes","2023","blog","LessWrong","www.lesswrong.com/posts/PJ7uvSB2qBjapxoWa/into-ai-safety-episodes-1-and-2",0,"",""],["Learning-theoretic agenda reading list","Vanessa Kosoy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fsGEyCYhqs7AWwdCe/learning-theoretic-agenda-reading-list",0,"","agents theory"],["Polysemantic Attention Head in a 4-Layer Transformer","Jett and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/nuJFTS5iiJKT5G5yh/polysemantic-attention-head-in-a-4-layer-transformer",0,"","interpretability mechanistic-interpretability"],["What we're missing: the case for structural risks from AI","Justin Olive","2023","blog","EA Forum","forum.effectivealtruism.org/posts/g3j7FfuHxFxWDGWpW/what-we-re-missing-the-case-for-structural-risks-from-ai",0,"","governance"],["​​ Open-ended/Phenomenal ​Ethics ​(TLTR)","Ryo","2023","blog","LessWrong","www.lesswrong.com/posts/iKLnEoYujBiGWvb5F/open-ended-phenomenal-ethics-tltr",0,"",""],["Alignment Frame/Exercise: Building The Puzzle of Alignment","Jonas Hallgren","2023","blog","LessWrong","www.lesswrong.com/posts/kGopQHxeKmJm3iXSe/alignment-frame-exercise-building-the-puzzle-of-alignment",0,"",""],["Growth and Form in a Toy Model of Superposition","Liam Carroll and Edmund Lau","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jvGqQGDrYzZM4MyaN/growth-and-form-in-a-toy-model-of-superposition",0,"","interpretability mechanistic-interpretability"],["Life of GPT","Odd anon","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wDaqyPJxhb6SASJSS/life-of-gpt",0,"",""],["Open-ended ethics of phenomena (a desiderata with universal morality)","Ryo","2023","blog","LessWrong","www.lesswrong.com/posts/K3m8K8JEweLZmGgv8/open-ended-ethics-of-phenomena-a-desiderata-with-universal",0,"",""],["Tall Tales at Different Scales: Evaluating Scaling Trends For Deception In Language Models","Felix Hofstätter and 5 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pip63HtEAxHGfSEGk/tall-tales-at-different-scales-evaluating-scaling-trends-for",0,"","evals deception scaling-laws"],["What’s going on? LLMs and IS-A sentences","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/BrKPfuPaqk8gyHciR/what-s-going-on-llms-and-is-a-sentences",0,"","interpretability"],["AI Alignment Research Engineer Accelerator (ARENA): call for applicants","TheMcDouglas and Kathryn O'Rourke","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LM2JnTHygKbn7eKLz/ai-alignment-research-engineer-accelerator-arena-call-for-1",0,"","interpretability"],["Announcing Athena - Women in AI Alignment Research","Claire Short","2023","blog","LessWrong","www.lesswrong.com/posts/pJ9qWeBRRuvPvnoNK/announcing-athena-women-in-ai-alignment-research",0,"",""],["Box inversion revisited","Jan_Kulveit","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jrKftFZMZjvNdQLNR/box-inversion-revisited",0,"","agents theory"],["Box inversion revisited","Jan_Kulveit","2023","blog","LessWrong","www.lesswrong.com/posts/jrKftFZMZjvNdQLNR/box-inversion-revisited",0,"","agents theory"],["Implementing Decision Theory","justinpombrio","2023","blog","LessWrong","www.lesswrong.com/posts/bqbL4pt92AGCCcvQH/implementing-decision-theory",0,"","theory"],["On the UK Summit","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/zbrvXGu264u3p8otD/on-the-uk-summit",0,"","governance"],["Please, someone make a dataset of supposed cases of \"tech panic\"","Harrison Durland","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6ercwC6JAPFTdKsy6/please-someone-make-a-dataset-of-supposed-cases-of-tech",0,"",""],["Scalable And Transferable Black-Box Jailbreaks For Language Models Via Persona Modulation","Soroush Pour and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kaxqjCKJL6RNHwJLD/scalable-and-transferable-black-box-jailbreaks-for-language-2",0,"","rlhf evals governance jailbreaks"],["Scalable And Transferable Black-Box Jailbreaks For Language Models Via Persona Modulation","soroushjp and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PutG2gC5huKK8ktWs/scalable-and-transferable-black-box-jailbreaks-for-language",0,"","evals governance policy jailbreaks"],["Scalable And Transferable Black-Box Jailbreaks For Language Models Via Persona Modulation","Soroush Pour and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/kaxqjCKJL6RNHwJLD/scalable-and-transferable-black-box-jailbreaks-for-language-2",0,"","rlhf evals governance jailbreaks"],["Thinking-in-limits about TAI from the demand perspective. Demand saturation, resource wars, new debt.","Ivan Madan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3KcYyn2qnRJ3LvSpn/thinking-in-limits-about-tai-from-the-demand-perspective",0,"","forecasting"],["20+ tips, tricks, lessons and thoughts on hosting hackathons","gergo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/f5wxYKFiJwjjRvk4Q/20-tips-tricks-lessons-and-thoughts-on-hosting-hackathons",0,"",""],["AI Fables Writing Contest Winners!","Daystar Eld","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EAmfYSBaJsMzHY2cW/ai-fables-writing-contest-winners",0,"",""],["An illustrative model of backfire risks from pausing AI research","Maxime Riché","2023","blog","LessWrong","www.lesswrong.com/posts/jjGiCZLuJ8ZNvZwQc/an-illustrative-model-of-backfire-risks-from-pausing-ai",0,"","forecasting"],["Announcing TAIS 2024","Blaine","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RdoFCykbfEyZzp675/announcing-tais-2024",0,"",""],["Governance of AI, Breakfast Cereal, Car Factories, Etc.","Jeff Martin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/APxdBEvcgGsmK5LAp/governance-of-ai-breakfast-cereal-car-factories-etc",0,"","governance policy"],["On running a city-wide university group","gergo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JALJeiqtZJbQfsMeq/on-running-a-city-wide-university-group",0,"",""],["Tips, tricks, lessons and thoughts on hosting hackathons","gergogaspar","2023","blog","LessWrong","www.lesswrong.com/posts/Abg5wRuBZrbkXZBFj/tips-tricks-lessons-and-thoughts-on-hosting-hackathons",0,"",""],["Why building ventures in AI Safety is particularly challenging","Heramb Podar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DhcaE7MbMwaCyNcxP/why-building-ventures-in-ai-safety-is-particularly",0,"",""],["Why building ventures in AI Safety is particularly challenging","Heramb","2023","blog","LessWrong","www.lesswrong.com/posts/ifLEKmhmk2utB64iX/why-building-ventures-in-ai-safety-is-particularly",0,"",""],["AI as Super-Demagogue","RationalDino","2023","blog","LessWrong","www.lesswrong.com/posts/DdDKsyA925Sm8BpQh/ai-as-super-demagogue",0,"","governance"],["Disentangling four motivations for acting in accordance with UDT","Julian Stastny","2023","blog","LessWrong","www.lesswrong.com/posts/nc2tzMLgXKc8NGzrQ/disentangling-four-motivations-for-acting-in-accordance-with",0,"","theory"],["Eric Schmidt on recursive self-improvement","nikola","2023","blog","LessWrong","www.lesswrong.com/posts/cLC2HcQbFZ5pFAgqC/eric-schmidt-on-recursive-self-improvement",0,"","forecasting"],["xAI announces Grok, beats GPT-3.5","nikola","2023","blog","LessWrong","www.lesswrong.com/posts/pbCw4QdL9K4tB7JEM/xai-announces-grok-beats-gpt-3-5",0,"","forecasting"],["[Linkpost] Concept Alignment as a Prerequisite for Value Alignment","Bogdan Ionut Cirstea","2023","blog","LessWrong","www.lesswrong.com/posts/M6CEJmgna6FTt9Yci/linkpost-concept-alignment-as-a-prerequisite-for-value",0,"",""],["Despair about AI progressing too slowly","Yarrow Bouchard","2023","blog","LessWrong","www.lesswrong.com/posts/pPrELFnR6Hp3vJWwQ/despair-about-ai-progressing-too-slowly",0,"","forecasting"],["Genetic fitness is a measure of selection strength, not the selection target","Kaj_Sotala","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BtffzD5yNB4CzSTJe/genetic-fitness-is-a-measure-of-selection-strength-not-the",0,"",""],["The 6D effect: When companies take risks, one email can be very powerful.","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/J9eF4nA6wJW6hPueN/the-6d-effect-when-companies-take-risks-one-email-can-be",0,"","governance"],["Untrusted smart models and trusted dumb models","Buck","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LhxHcASQwpNa3mRNk/untrusted-smart-models-and-trusted-dumb-models",0,"",""],["We are already in a persuasion-transformed world and must take precautions","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/LdEwDn5veAckEemi4/we-are-already-in-a-persuasion-transformed-world-and-must",0,"","governance"],["If AGI is imminent, why can’t I hail a robotaxi?","Yarrow Bouchard","2023","blog","LessWrong","www.lesswrong.com/posts/cyycbDAffNc6aghas/if-agi-is-imminent-why-can-t-i-hail-a-robotaxi",0,"","forecasting"],["Paul Christiano on Dwarkesh Podcast","ESRogs","2023","blog","LessWrong","www.lesswrong.com/posts/7FrHeyxQpb3pvr9Pn/paul-christiano-on-dwarkesh-podcast",0,"","rlhf alignment-faking deception forecasting"],["Sam Altman: \"safety and capabilities are not these two separate things\"","Yarrow Bouchard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vuATadXMheRhBvXfi/sam-altman-safety-and-capabilities-are-not-these-two",0,"",""],["The Navigation Fund launched + is hiring a program officer to lead the distribution of $20M annually for AI safety! Full-time, fully remote, pay starts at $200k","vincentweisser","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NAcN98bACuwcnB32H/the-navigation-fund-launched-is-hiring-a-program-officer-to",0,"",""],["Thoughts on open source AI","Sam Marks","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WLYBy5Cus4oRFY3mu/thoughts-on-open-source-ai",0,"",""],["Why is learning economics, psychology, sociology important for preventing AI risks?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/P9bPprknW9pApdKvg/why-is-learning-economics-psychology-sociology-important-for",0,"",""],["A Critique of The Evidentialist's Wager","Heighn","2023","blog","LessWrong","www.lesswrong.com/posts/npg4AkbvwhkDay5jX/a-critique-of-the-evidentialist-s-wager",0,"","theory"],["Mech Interp Challenge: November - Deciphering the Cumulative Sum Model","TheMcDouglas","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uPa63suC8idWhYGbg/mech-interp-challenge-november-deciphering-the-cumulative",0,"","interpretability"],["Still no strong evidence that LLMs increase bioterrorism risk","freedomandutility","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zLkdQRFBeyyMLKoNj/still-no-strong-evidence-that-llms-increase-bioterrorism",0,"","policy"],["[Congressional Hearing] Oversight of A.I.: Legislating on Artificial Intelligence","Tristan Williams","2023","blog","EA Forum","forum.effectivealtruism.org/posts/r8kZ78uBs6XTMhKes/congressional-hearing-oversight-of-a-i-legislating-on",0,"","policy"],["AI Alignment: A Comprehensive Survey","Stephen McAleer","2023","blog","LessWrong","www.lesswrong.com/posts/sL9qmAqgB2RL6JFca/ai-alignment-a-comprehensive-survey",0,"","alignment-faking deception"],["Dario Amodei’s prepared remarks from the UK AI Safety Summit, on Anthropic’s Responsible Scaling Policy","Zac Hatfield-Dodds","2023","blog","LessWrong","www.lesswrong.com/posts/vm7FRyPWGCqDHy6LF/dario-amodei-s-prepared-remarks-from-the-uk-ai-safety-summit",0,"","governance policy"],["Forecasting Questions: What do you want to predict on AI?","Nathan Young","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oJ6jAhbbPADGASpLn/forecasting-questions-what-do-you-want-to-predict-on-ai",0,"","forecasting"],["My thoughts on the social response to AI risk","Matthew Barnett","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EaZghEwcCJRAuee66/my-thoughts-on-the-social-response-to-ai-risk",0,"",""],["On the Executive Order","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/PvBpRu354uG7ypwRP/on-the-executive-order",0,"","governance"],["Reactions to the Executive Order","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/G8SsspgAYEHHiDGNP/reactions-to-the-executive-order",0,"","governance"],["Robustness of Contrast-Consistent Search to Adversarial Prompting","Nandi and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/zZbM5JdMs5uCtMkgs/robustness-of-contrast-consistent-search-to-adversarial",0,"","interpretability eliciting-latent-knowledge robustness"],["Singular learning theory and bridging from ML to brain emulations","kave and Garrett Baker","2023","blog","LessWrong","www.lesswrong.com/posts/PikpeRucdsXeEvpy9/singular-learning-theory-and-bridging-from-ml-to-brain",0,"",""],["Snapshot of narratives and frames against regulating AI","Jan_Kulveit","2023","blog","LessWrong","www.lesswrong.com/posts/PdcnEEE6sdgACDrEk/snapshot-of-narratives-and-frames-against-regulating-ai",0,"","governance"],["The Bletchley Declaration on AI Safety","Hauke Hillebrandt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/j2TreuRZT9mBFEMEs/the-bletchley-declaration-on-ai-safety",0,"",""],["Agent Foundations track in MATS","Vanessa Kosoy","2023","blog","LessWrong","www.lesswrong.com/posts/ndSQYCsXJmbsywzZ7/agent-foundations-track-in-mats",0,"","agents theory"],["AI Safety 101 - Chapter 5.1 - Debate","Charbel-Raphaël","2023","blog","LessWrong","www.lesswrong.com/posts/WP4fciGn3rNtmq3tY/ai-safety-101-chapter-5-1-debate",0,"",""],["AISN #25: White House Executive Order on AI, UK AI Safety Summit, and Progress on Voluntary Evaluations of AI Risks","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SH3Es2Q9XuYPMhA5H/aisn-25-white-house-executive-order-on-ai-uk-ai-safety",0,"","evals"],["AISN #25: White House Executive Order on AI, UK AI Safety Summit, and Progress on Voluntary Evaluations of AI Risks","aogara and Dan H","2023","blog","LessWrong","www.lesswrong.com/posts/ARK4tyCRDd3Gb5umh/aisn-25-white-house-executive-order-on-ai-uk-ai-safety",0,"","evals"],["Preventing Language Models from hiding their reasoning","Fabien Roger and ryan_greenblatt","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9Fdd9N7Escg3tcymb/preventing-language-models-from-hiding-their-reasoning",0,"","evals"],["The UK AI Safety Summit tomorrow","SebastianSchmidt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2sZudkyLtNqsuskE5/the-uk-ai-safety-summit-tomorrow",0,"","policy"],["Thoughts on the AI Safety Summit company policy requests and responses","So8res","2023","blog","LessWrong","www.lesswrong.com/posts/ms3x8ngwTfep7jBue/thoughts-on-the-ai-safety-summit-company-policy-requests-and",0,"","governance policy"],["Urging an International AI Treaty: An Open Letter","Loppukilpailija","2023","blog","LessWrong","www.lesswrong.com/posts/epC5CrGCv4JGdfsjm/urging-an-international-ai-treaty-an-open-letter",0,"","governance compute-governance"],["5 Reasons Why Governments/Militaries Already Want AI for Information Warfare","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/jyAerr8txxhiKnxwA/5-reasons-why-governments-militaries-already-want-ai-for",0,"","governance"],["[Linkpost] Two major announcements in AI governance today","anonymous","2023","blog","LessWrong","www.lesswrong.com/posts/XNqCRLtc2syiDbQYn/linkpost-two-major-announcements-in-ai-governance-today",0,"","governance"],["Charbel-Raphaël and Lucius discuss Interpretability","Mateusz Bagiński and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/FDrgcfY8zs5e2eJDd/charbel-raphael-and-lucius-discuss-interpretability",0,"","interpretability"],["M&A in AI","Hauke Hillebrandt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zozxDjHkizsfWLEC3/m-and-a-in-ai",0,"",""],["President Biden Issues Executive Order on Safe, Secure, and Trustworthy Artificial Intelligence","Tristan Williams","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pcbsM45vLmHcFpNnr/president-biden-issues-executive-order-on-safe-secure-and",0,"","policy"],["Response to “Coordinated pausing: An evaluation-based coordination scheme for frontier AI developers”","Matthew Wearden","2023","blog","LessWrong","www.lesswrong.com/posts/JCGAdrrr3ePXHEzqc/response-to-coordinated-pausing-an-evaluation-based",0,"","evals governance"],["Will releasing the weights of large language models grant widespread access to pandemic agents?","Jeff Kaufman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZuzK2s4JsJcexBJxy/will-releasing-the-weights-of-large-language-models-grant",0,"","agents"],["Would it make sense to bring a civil lawsuit against Meta for recklessly open sourcing models?","Nathan Helm-Burger","2023","blog","LessWrong","www.lesswrong.com/posts/dL3qxebM29WjwtSAv/would-it-make-sense-to-bring-a-civil-lawsuit-against-meta",0,"","governance"],["Clarifying the free energy principle (with quotes)","Ryo","2023","blog","LessWrong","www.lesswrong.com/posts/bebw3SEjXY3SCAcwD/clarifying-the-free-energy-principle-with-quotes",0,"","theory"],["The AI Boom Mainly Benefits Big Firms, but long-term, markets will concentrate","Hauke Hillebrandt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9FPvJ4dXeiXYKwFpL/the-ai-boom-mainly-benefits-big-firms-but-long-term-markets",0,"","forecasting"],["Regrant up to $600,000 to AI safety projects with GiveWiki","Dawn Drescher","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zxxew56gnYhYEupsc/regrant-up-to-usd600-000-to-ai-safety-projects-with-givewiki",0,"",""],["Summary: Existential risk from power-seeking AI by Joseph Carlsmith","rileyharris","2023","blog","EA Forum","forum.effectivealtruism.org/posts/caqjHNvAQc6B8auHM/summary-existential-risk-from-power-seeking-ai-by-joseph",0,"","power-seeking"],["AI safety field-building survey: Talent needs, infrastructure needs, and relationship to EA","michel and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vzuqnPyfDFjtbCpgv/ai-safety-field-building-survey-talent-needs-infrastructure",0,"",""],["Efficacy of AI Activism: Have We Ever Said No?","charlieh943","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WfodoyjePTTuaTjLe/efficacy-of-ai-activism-have-we-ever-said-no",0,"","governance"],["Linkpost: Rishi Sunak's Speech on AI (26th October)","bideup","2023","blog","LessWrong","www.lesswrong.com/posts/94nYiPnr34kmHLMrB/linkpost-rishi-sunak-s-speech-on-ai-26th-october",0,"","governance"],["New report on the state of AI safety in China","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tkXPqvpCGaeNqBgSe/new-report-on-the-state-of-ai-safety-in-china",0,"","policy"],["Value systematization: how values become coherent (and misaligned)","Richard_Ngo","2023","blog","LessWrong","www.lesswrong.com/posts/J2kpxLjEyqh6x3oA4/value-systematization-how-values-become-coherent-and",0,"",""],["We're Not Ready: thoughts on \"pausing\" and responsible scaling policies","HoldenKarnofsky","2023","blog","LessWrong","www.lesswrong.com/posts/Np5Q3Mhz2AiPtejGN/we-re-not-ready-thoughts-on-pausing-and-responsible-scaling-4",0,"","governance"],["Wireheading and misalignment by composition on NetHack","pierlucadoro","2023","blog","LessWrong","www.lesswrong.com/posts/GEjzyf7Hjpv9g2uGX/wireheading-and-misalignment-by-composition-on-nethack",0,"","rlhf reward-hacking"],["1. Premise one: Values are malleable","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HyodRjYtiA2xozCrk/1-premise-one-values-are-malleable",0,"",""],["2. Premise two: Some cases of value change are (il)legitimate","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QjA6kipHYqwACkPNw/2-premise-two-some-cases-of-value-change-are-il-legitimate",0,"",""],["3. Premise three & Conclusion: AI systems can affect value change trajectories & the Value Change Problem","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yPnAzeRAqdko3RNtR/3-premise-three-and-conclusion-ai-systems-can-affect-value",0,"",""],["4. Risks from causing illegitimate value change (performative predictors)","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qZFGPJi3u8xuvnWHQ/4-risks-from-causing-illegitimate-value-change-performative",0,"",""],["5. Risks from preventing legitimate value change (value collapse)","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KeHGinpj2WyzDEQAx/5-risks-from-preventing-legitimate-value-change-value",0,"",""],["AI #35: Responsible Scaling Policies","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/aQ6LDhc2zxrYXFjEF/ai-35-responsible-scaling-policies",0,"","governance"],["Apply to the Constellation Visiting Researcher Program and Astra Fellowship, in Berkeley this Winter","Nate Thomas","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/oBdfDvmrBKoTq3x85/apply-to-the-constellation-visiting-researcher-program-and",0,"",""],["Apply to the Constellation Visiting Researcher Program and Astra Fellowship, in Berkeley this Winter","Anjay F and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fkPBQNNuDzSeX8jmp/apply-to-the-constellation-visiting-researcher-program-and",0,"",""],["CHAI internship applications are open (due Nov 13)","Erik Jenner","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gnoy5dvaJg3NvqPHz/chai-internship-applications-are-open-due-nov-13",0,"",""],["Codebook Features: Sparse and Discrete Interpretability for Neural Networks","Alex Tamkin and 2 others","2023","paper","arXiv preprint","arxiv.org/abs/2310.17230",0,"","interpretability mechanistic-interpretability"],["Disagreements over the prioritization of existential risk from AI","Olivier Coutu","2023","blog","LessWrong","www.lesswrong.com/posts/eAT2dXAngXxFRTQLn/disagreements-over-the-prioritization-of-existential-risk",0,"","governance"],["OpenAI’s new Preparedness team is hiring","leopold","2023","blog","EA Forum","forum.effectivealtruism.org/posts/k2iaSvbnpQFzE4nLB/openai-s-new-preparedness-team-is-hiring",0,"","evals forecasting"],["UK Government publishes \"Frontier AI: capabilities and risks\" Discussion Paper","A.H.","2023","blog","LessWrong","www.lesswrong.com/posts/eZ8xAyxiELASGsawb/uk-government-publishes-frontier-ai-capabilities-and-risks",0,"",""],["UK Prime Minister Rishi Sunak's Speech on AI","Tobias Häberli","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3qpaRKe8R4ptiqSkr/uk-prime-minister-rishi-sunak-s-speech-on-ai",0,"","policy"],["What we learned from running an Australian AI Safety Unconference","Alexander Saeri and Jo Small","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rBteAqbfqvaFMvpv5/what-we-learned-from-running-an-australian-ai-safety",0,"",""],["AI as a science, and three obstacles to alignment strategies","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JcLhYQQADzTsAEaXd/ai-as-a-science-and-three-obstacles-to-alignment-strategies",0,"",""],["Announcing Epoch's newly expanded Parameters, Compute and Data Trends in Machine Learning database","Robi Rahman and Jaime Sevilla","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pAHPpX4cAwjtkLYkT/announcing-epoch-s-newly-expanded-parameters-compute-and",0,"","forecasting"],["Anthropic, Google, Microsoft & OpenAI announce Executive Director of the Frontier Model Forum & over $10 million for a new AI Safety Fund","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/5jpESFymqEgSAmDJL/anthropic-google-microsoft-and-openai-announce-executive",0,"","governance"],["Compositional preference models for aligning LMs","Tomek Korbak","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/oSgac8x8fgNj22ky3/compositional-preference-models-for-aligning-lms",0,"","rlhf"],["Compositional preference models for aligning LMs","Tomek Korbak","2023","blog","LessWrong","www.lesswrong.com/posts/oSgac8x8fgNj22ky3/compositional-preference-models-for-aligning-lms",0,"","rlhf"],["Responsible Scaling Policies Are Risk Management Done Wrong","simeon_c","2023","blog","LessWrong","www.lesswrong.com/posts/9nEBWxjAHSu3ncr6v/responsible-scaling-policies-are-risk-management-done-wrong",0,"","governance"],["Successif: Join our AI program to help mitigate the catastrophic risks of AI","ClaireB and AzrielZ","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nbYRmenLjF3wE45sm/successif-join-our-ai-program-to-help-mitigate-the",0,"","governance"],["#168 – Whether deep history says we’re heading for an intelligence explosion (Ian Morris on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7WvoCbWfa6kWsgbA9/168-whether-deep-history-says-we-re-heading-for-an",0,"",""],["[Interview w/ Quintin Pope] Evolution, values, and AI Safety","fowlertm","2023","blog","LessWrong","www.lesswrong.com/posts/c9W4SHa7DHwAHkqRF/interview-w-quintin-pope-evolution-values-and-ai-safety",0,"",""],["Announcing #AISummitTalks featuring Professor Stuart Russell and many others","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XFGdTab6eMJriDMGD/announcing-aisummittalks-featuring-professor-stuart-russell",0,"","governance policy"],["Announcing #AISummitTalks featuring Professor Stuart Russell and many others","otto.barten","2023","blog","LessWrong","www.lesswrong.com/posts/NaXz3FM9gXXB7oJW3/announcing-aisummittalks-featuring-professor-stuart-russell",0,"","governance forecasting"],["Go Mobilize? Lessons from GM Protests for Pausing AI","charlieh943","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6jxrzk99eEjsBxoMA/go-mobilize-lessons-from-gm-protests-for-pausing-ai",0,"","governance"],["Human wanting","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YLRPhvgN4uZ6LCLxw/human-wanting",0,"",""],["Largest AI model in 2 years from $10B","Péter Drótos","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aoBxSba4CsEAtHqRy/largest-ai-model-in-2-years-from-usd10b",0,"","governance compute-governance forecasting"],["Lying is Cowardice, not Strategy","Connor Leahy and Gabriel Alfour","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qtTW6BFrxWw4iHcjf/lying-is-cowardice-not-strategy",0,"","deception"],["The Self in Artificial Consciousness: A Buddhist Investigation into Advanced AI","Ryan Combes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BE3cdKEaLKsnHNqim/the-self-in-artificial-consciousness-a-buddhist",0,"",""],["Thoughts on responsible scaling policies and regulation","paulfchristiano","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dxgEaDrEBkkE96CXr/thoughts-on-responsible-scaling-policies-and-regulation",0,"","governance"],["Thoughts on responsible scaling policies and regulation","Paul_Christiano","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cKW4db8u2uFEAHewg/thoughts-on-responsible-scaling-policies-and-regulation",0,"","governance"],["Towards Understanding Sycophancy in Language Models","Ethan Perez and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/g5rABd5qbp8B4g3DE/towards-understanding-sycophancy-in-language-models",0,"","rlhf sycophancy"],["Towards Understanding Sycophancy in Language Models","Ethan Perez and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/g5rABd5qbp8B4g3DE/towards-understanding-sycophancy-in-language-models",0,"","rlhf sycophancy"],["Who is Harry Potter? Some predictions.","Donald Hobson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/B4vgbeXMGxEnEwY8d/who-is-harry-potter-some-predictions",0,"",""],["Fundamental Challenges in AI Governance","Tharin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GnALeFmKbknkGYdp8/fundamental-challenges-in-ai-governance",0,"","governance policy"],["Help us design the interface for aisafety.com","Kim Holder","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PMhKbnaky7hMopWhM/help-us-design-the-interface-for-aisafety-com",0,"",""],["Machine Unlearning Evaluations as Interpretability Benchmarks","NickyP and Nandi","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mTi8TQEyP5Pr7oczd/machine-unlearning-evaluations-as-interpretability",0,"","interpretability evals benchmarks unlearning"],["Machine Unlearning Evaluations as Interpretability Benchmarks","NickyP and Nandi","2023","blog","LessWrong","www.lesswrong.com/posts/mTi8TQEyP5Pr7oczd/machine-unlearning-evaluations-as-interpretability",0,"","interpretability evals benchmarks unlearning"],["Open Source Replication & Commentary on Anthropic's Dictionary Learning Paper","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fKuugaxt2XLTkASkk/open-source-replication-and-commentary-on-anthropic-s",0,"","mechanistic-interpretability"],["Pausing AI might be good policy, but it's bad politics","Stephen Clare","2023","blog","EA Forum","forum.effectivealtruism.org/posts/avrFeH6LpqJrjmGmc/pausing-ai-might-be-good-policy-but-it-s-bad-politics",0,"","policy robustness"],["Programmatic backdoors: DNNs can use SGD to run arbitrary stateful computation","Fabien Roger and Buck","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QNQuWB3hS5FrGp5yZ/programmatic-backdoors-dnns-can-use-sgd-to-run-arbitrary",0,"",""],["The Shutdown Problem: Three Theorems","EJT","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zb22pAKoFGsqKwnCg/the-shutdown-problem-three-theorems",0,"",""],["VLM-RM: Specifying Rewards with Natural Language","ChengCheng and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vyxwgQnWPdhpWQ9ZN/vlm-rm-specifying-rewards-with-natural-language",0,"","rlhf"],["VLM-RM: Specifying Rewards with Natural Language","ChengCheng and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/vyxwgQnWPdhpWQ9ZN/vlm-rm-specifying-rewards-with-natural-language",0,"","rlhf"],["Announcing Timaeus","Jesse Hoogland and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nN7bHuHZYaWv9RDJL/announcing-timaeus",0,"","interpretability"],["Announcing Timaeus","Stan van Wingerden and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aaa9pEwnvyeE2TYvg/announcing-timaeus",0,"","interpretability"],["Into AI Safety - Episode 0","jacobhaimes","2023","blog","LessWrong","www.lesswrong.com/posts/kKFTzraz9oyajkcEH/into-ai-safety-episode-0",0,"",""],["AI Safety is Dropping the Ball on Clown Attacks, and Mind Control in General","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/mjSjPHCtbK6TA5tfW/ai-safety-is-dropping-the-ball-on-clown-attacks-and-mind",0,"",""],["Alignment Implications of LLM Successes: a Debate in One Act","Zack_M_Davis","2023","blog","LessWrong","www.lesswrong.com/posts/pYWA7hYJmXnuyby33/alignment-implications-of-llm-successes-a-debate-in-one-act",0,"",""],["Apply for MATS Winter 2023-24!","Rocket and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WdvrgRKLfYQRw5bRD/apply-for-mats-winter-2023-24",0,"",""],["Apply for MATS Winter 2023-24!","Rocket and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/tqyg3DpoiE4DKyi4y/apply-for-mats-winter-2023-24",0,"",""],["How toy models of ontology changes can be misleading","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/XvhrmTog2bkf5s2qu/how-toy-models-of-ontology-changes-can-be-misleading",0,"",""],["Thoughts On (Solving) Deep Deception","Jozdien","2023","blog","LessWrong","www.lesswrong.com/posts/Lm8vTwXdDMEojR85A/thoughts-on-solving-deep-deception-1",0,"","interpretability alignment-faking deception"],["Announcing new round of \"Key Phenomena in AI Risk\" Reading Group","DusanDNesic and Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vakhhNHduW9gmENTW/announcing-new-round-of-key-phenomena-in-ai-risk-reading",0,"",""],["I Would Have Solved Alignment, But I Was Worried That Would Advance Timelines","307th","2023","blog","LessWrong","www.lesswrong.com/posts/Eu8y4cTxM3pAzwdCf/i-would-have-solved-alignment-but-i-was-worried-that-would",0,"","forecasting"],["Internal Target Information for AI Oversight","Paul Colognese","2023","blog","LessWrong","www.lesswrong.com/posts/hhKpXEsfAiyFLecyF/internal-target-information-for-ai-oversight",0,"","interpretability"],["Revealing Intentionality In Language Models Through AdaVAE Guided Sampling","jdp","2023","blog","LessWrong","www.lesswrong.com/posts/4Hnso8NMAeeYs8Cta/revealing-intentionality-in-language-models-through-adavae",0,"","interpretability situational-awareness"],["Specific versus General Principles for Constitutional AI","Sandipan Kundu and 29 others","2023","paper","arXiv preprint","arxiv.org/abs/2310.13798",0,"","rlhf constitutional-ai"],["TOMORROW: the largest AI Safety protest ever!","Holly_Elmore","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WZR5ZmC9nvFXeaySS/tomorrow-the-largest-ai-safety-protest-ever",0,"",""],["Towards Understanding Sycophancy in Language Models","Mrinank Sharma","2023","paper","arXiv preprint","arxiv.org/abs/2310.13548",0,"","rlhf evals sycophancy"],["Guess, ask or tell?","dEAsign","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tBLsC2jZYxLYrCdbN/guess-ask-or-tell",0,"",""],["New roles on my team: come build Open Phil's technical AI safety program with me!","Ajeya","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SQSXfiByKat2YzpWu/new-roles-on-my-team-come-build-open-phil-s-technical-ai",0,"",""],["Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","Juan Rocamonde","2023","paper","arXiv preprint","arxiv.org/abs/2310.12921",0,"","rlhf agents"],["(Non-deceptive) Suboptimality Alignment","Sodium","2023","blog","LessWrong","www.lesswrong.com/posts/h8GTzLBAb4oRKgKbM/non-deceptive-suboptimality-alignment",0,"","deception"],["AISN #24: Kissinger Urges US-China Cooperation on AI, China's New AI Law, US Export Controls, International Institutions, and Open Source AI","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SvjiueLQpjJLRehuF/aisn-24-kissinger-urges-us-china-cooperation-on-ai-china-s",0,"","policy"],["Alignment 101 - Ch.1 - AGI","markov","2023","blog","LessWrong","www.lesswrong.com/posts/YcFpJC5pJFdYdEuNN/alignment-101-ch-1-agi",0,"",""],["Alignment 101 - Ch.2 - Reward Misspecification","markov","2023","blog","LessWrong","www.lesswrong.com/posts/mMBoPnFrFqQJKzDsZ/alignment-101-ch-2-reward-misspecification",0,"",""],["Metaculus Launches Conditional Cup to Explore Linked Forecasts","christian","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JMhwY2WE3oqkRxf6h/metaculus-launches-conditional-cup-to-explore-linked",0,"","forecasting"],["On Interpretability's Robustness","WCargo","2023","blog","LessWrong","www.lesswrong.com/posts/tNdSqrk6hpxfxmZqS/on-interpretability-s-robustness",0,"","interpretability robustness"],["The (partial) fallacy of dumb superintelligence","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/qsDPHZwjmduSMCJLv/the-partial-fallacy-of-dumb-superintelligence",0,"",""],["Beginner’s guide to reducing s-risks [link-post]","Center on Long-Term Risk","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MymQqnT8gZ2yjmeYX/beginner-s-guide-to-reducing-s-risks-link-post",0,"",""],["Investigating the learning coefficient of modular addition: hackathon project","Nina Rimsky and Dmitry Vaintrob","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4v3hMuKfsGatLXPgt/investigating-the-learning-coefficient-of-modular-addition-1",0,"",""],["The theoretical computational limit of the Solar System is 1.47x10^49 bits per second.","William the Kiwi","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XHedou8TjeAccuerm/the-theoretical-computational-limit-of-the-solar-system-is-1",0,"","forecasting"],["Goodhart's Law in Reinforcement Learning","jacek and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/Eu6CvP7c7ivcGM3PJ/goodhart-s-law-in-reinforcement-learning",0,"","goodharts-law"],["Knowledge Base 4: General applications","iwis","2023","blog","LessWrong","www.lesswrong.com/posts/8e9HsZsw8QuRwnqLX/knowledge-base-4-general-applications",0,"",""],["Neuronpedia - AI Safety Game","johnnylin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/syEQKdmhNHbrBqtwe/neuronpedia-ai-safety-game",0,"",""],["UNGA General Debate speeches on AI","Odd anon","2023","blog","LessWrong","www.lesswrong.com/posts/x3JpgTnqcrzhedwAb/unga-general-debate-speeches-on-ai",0,"","governance"],["Discovering Latent Knowledge in the Human Brain: Part 1 – Clarifying the concepts of belief and knowledge","Joseph Emerson","2023","blog","LessWrong","www.lesswrong.com/posts/csFXHGb7gxpzMTeT5/discovering-latent-knowledge-in-the-human-brain-part-1",0,"","eliciting-latent-knowledge"],["Mapping ChatGPT’s ontological landscape, gradients and choices [interpretability]","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/renezm5cFCuMBBv9s/mapping-chatgpt-s-ontological-landscape-gradients-and",0,"","interpretability"],["Politico article on Open Phil, Horizon Fellowship, and EA","Calum","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uiyHiwrXKysfdoCps/politico-article-on-open-phil-horizon-fellowship-and-ea",0,"",""],["Assessing the Dangerousness of Malevolent Actors in AGI Governance: A Preliminary Exploration","Callum Hinchcliffe and RichardAnnilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dreEnpSSohfkmZdCB/assessing-the-dangerousness-of-malevolent-actors-in-agi",0,"","governance"],["ChatGPT tells 20 versions of its prototypical story, with a short note on method","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/LK8R8YmndScXjeynx/chatgpt-tells-20-versions-of-its-prototypical-story-with-a",0,"","interpretability"],["Natural Abstraction: Convergent Preferences Over Information Structures","paulom","2023","blog","LessWrong","www.lesswrong.com/posts/gts62zc6roEWzZEsg/natural-abstraction-convergent-preferences-over-information-4",0,"","instrumental-convergence power-seeking"],["RSPs are pauses done right","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mcnWZBnbeDz7KKtjJ/rsps-are-pauses-done-right",0,"",""],["Which Anaesthetic To Choose?","dadadarren","2023","blog","LessWrong","www.lesswrong.com/posts/TkgZWZKXgcCLc3G55/which-anaesthetic-to-choose",0,"","theory"],["[Paper] All's Fair In Love And Love: Copy Suppression in GPT-2 Small","TheMcDouglas and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/ebezsHW6qJwxTFasX/paper-all-s-fair-in-love-and-love-copy-suppression-in-gpt-2",0,"","interpretability"],["At Our World in Data we're hiring our first Communications & Outreach Manager","Charlie Giattino","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nov4gS6iRFgvuPu2m/at-our-world-in-data-we-re-hiring-our-first-communications",0,"","policy"],["FLI podcast series, \"Imagine A World\", about aspirational futures with AGI","Jackson Wagner","2023","blog","LessWrong","www.lesswrong.com/posts/zaFwokgn9MxtYd46E/fli-podcast-series-imagine-a-world-about-aspirational",0,"","governance"],["How can I best use my career to pass impactful AI and Biosecurity policy.","maxg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SCiAvwbczwag8on6S/how-can-i-best-use-my-career-to-pass-impactful-ai-and",0,"","policy"],["Paper: Understanding and Controlling a Maze-Solving Policy Network","TurnTrout and 6 others","2023","blog","LessWrong","www.lesswrong.com/posts/DKtWikjcdApRj3rWr/paper-understanding-and-controlling-a-maze-solving-policy",0,"","interpretability policy"],["To open-source or to not open-source, that is (an oversimplification of) the question.","Justin Bullock","2023","blog","LessWrong","www.lesswrong.com/posts/9HeZjGpkQQJfkcbqh/to-open-source-or-to-not-open-source-that-is-an",0,"","governance"],["What he’s learned as an AI policy insider (Tantum Collins on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YRjsjis8LDFwg2btr/what-he-s-learned-as-an-ai-policy-insider-tantum-collins-on",0,"","policy"],["2024 S-risk Intro Fellowship","Center on Long-Term Risk","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rykCowkpDJiwr9t2G/2024-s-risk-intro-fellowship",0,"",""],["Looking for reading recommendations: Theories of right/justice that safeguard against having one's job automated?","bulKlub","2023","blog","LessWrong","www.lesswrong.com/posts/DKwaWnsE4LFss439N/looking-for-reading-recommendations-theories-of-right",0,"","governance"],["LoRA Fine-tuning Efficiently Undoes Safety Training from Llama 2-Chat 70B","Simon Lermen and Jeffrey Ladish","2023","blog","LessWrong","www.lesswrong.com/posts/qmQFHCgCyEEjuy5a7/lora-fine-tuning-efficiently-undoes-safety-training-from",0,"","rlhf"],["Opportunities for Impact Beyond the EU AI Act","Cillian_","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pvDGtDbaSj8gZuwN5/opportunities-for-impact-beyond-the-eu-ai-act",0,"","governance policy"],["Relevance of 'Harmful Intelligence' Data in Training Datasets (WebText vs. Pile)","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/Bm5QhjiWs95YL4Kgt/relevance-of-harmful-intelligence-data-in-training-datasets",0,"","training-data"],["Resources & opportunities for careers in European AI Policy","Cillian_ and Training for Good","2023","blog","EA Forum","forum.effectivealtruism.org/posts/W4BRXGvz7BvMPFNvy/resources-and-opportunities-for-careers-in-european-ai",0,"","governance policy"],["The International PauseAI Protest: Activism under uncertainty","Joseph Miller and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eqTGrEsBzJJSiuTcv/the-international-pauseai-protest-activism-under-uncertainty",0,"","governance"],["Timelines are short, p(doom) is high: a global stop to frontier AI development until x-safety consensus is our only reasonable hope","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/E6CahapSad7psvqx4/timelines-are-short-p-doom-is-high-a-global-stop-to-frontier",0,"","forecasting"],["unRLHF - Efficiently undoing LLM safeguards","Pranav Gade and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/3eqHYxfWb5x4Qfz8C/unrlhf-efficiently-undoing-llm-safeguards",0,"","rlhf governance"],["Attributing to interactions with GCPD and GWPD","jenny","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hFaXe4Mi64xkE6Kqp/attributing-to-interactions-with-gcpd-and-gwpd",0,"","interpretability eliciting-latent-knowledge"],["Attributing to interactions with GCPD and GWPD","jenny","2023","blog","LessWrong","www.lesswrong.com/posts/hFaXe4Mi64xkE6Kqp/attributing-to-interactions-with-gcpd-and-gwpd",0,"","interpretability eliciting-latent-knowledge"],["Understanding LLMs: Some basic observations about words, syntax, and discourse [w/ a conjecture about grokking]","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/NwRpyRS3LRk4yXe8T/understanding-llms-some-basic-observations-about-words",0,"","interpretability"],["Update on the UK AI Taskforce & AI Safety Summit","Elliot_Mckernon","2023","blog","LessWrong","www.lesswrong.com/posts/5fdcsWwtvG9jAtzGK/update-on-the-uk-ai-taskforce-and-ai-safety-summit",0,"","governance"],["You’re Measuring Model Complexity Wrong","Jesse Hoogland and Stan van Wingerden","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6g8cAftfQufLmFDYT/you-re-measuring-model-complexity-wrong",0,"","interpretability"],["A New Model for Compute Center Verification","Damin Curtis","2023","blog","LessWrong","www.lesswrong.com/posts/qAjGfYh2pQvvnsCBk/a-new-model-for-compute-center-verification-1",0,"","governance compute-governance"],["AI+bio cannot be half of AI catastrophe risk, right?","Ulrik Horn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ARwFCMpgTbmJ89hBP/ai-bio-cannot-be-half-of-ai-catastrophe-risk-right",0,"",""],["Become a PIBBSS Research Affiliate","Nora_Ammann and DusanDNesic","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JvL3SC6tPjFyCiHad/become-a-pibbss-research-affiliate-1",0,"",""],["Documenting Journey Into AI Safety","jacobhaimes","2023","blog","LessWrong","www.lesswrong.com/posts/ozDWnEChJwuB5L5wg/documenting-journey-into-ai-safety",0,"",""],["Non-superintelligent paperclip maximizers are normal","jessicata","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Z8C29oMAmYjhk2CNN/non-superintelligent-paperclip-maximizers-are-normal",0,"",""],["Pause For Thought: The AI Pause Debate","Scott Alexander","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7WfMYzLfcTyDtD6Gn/pause-for-thought-the-ai-pause-debate",0,"","governance"],["Scale, schlep, and systems","Ajeya","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4MkkdbSa42h73pXi8/scale-schlep-and-systems",0,"",""],["The Bostrom Buckle: Visualising the Vulnerable World Hypothesis","Rosco-Hunter","2023","blog","LessWrong","www.lesswrong.com/posts/Cs8FaAxYHGgzqpSkh/the-bostrom-buckle-visualising-the-vulnerable-world",0,"","governance"],["We don't understand what happened with culture enough","Jan_Kulveit","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wCtegGaWxttfKZsfx/we-don-t-understand-what-happened-with-culture-enough",0,"","forecasting"],["We don't understand what happened with culture enough","Jan_Kulveit","2023","blog","LessWrong","www.lesswrong.com/posts/wCtegGaWxttfKZsfx/we-don-t-understand-what-happened-with-culture-enough",0,"","forecasting"],["Perspective Based Reasoning Could Absolve CDT","dadadarren","2023","blog","LessWrong","www.lesswrong.com/posts/izNiFpyWgqddTz34t/perspective-based-reasoning-could-absolve-cdt-1",0,"","theory"],["Silicon Valley’s Rabbit Hole Problem","Mandelbrot","2023","blog","EA Forum","forum.effectivealtruism.org/posts/R9GbQQksznh2SwS4y/silicon-valley-s-rabbit-hole-problem",0,"",""],["Time is homogeneous sequentially-composable determination","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6qrLfAG7mDTyrHmh7/time-is-homogeneous-sequentially-composable-determination",0,"",""],["Comparing Anthropic's Dictionary Learning to Ours","Robert_AIZI","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/F4iogK5xdNd7jDNyw/comparing-anthropic-s-dictionary-learning-to-ours",0,"","interpretability mechanistic-interpretability"],["Don't Dismiss Simple Alignment Approaches","Chris_Leong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ziNCZEm7FE9LHxLai/don-t-dismiss-simple-alignment-approaches",0,"",""],["Fixing Insider Threats in the AI Supply Chain","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/C5X3XbHQkj5d8EeXg/fixing-insider-threats-in-the-ai-supply-chain",0,"",""],["Risk-averse Batch Active Inverse Reward Design","Panagiotis Liampas","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JKwRejsticvZg2vre/risk-averse-batch-active-inverse-reward-design",0,"",""],["Utilitarianism is irrational or self-undermining","MichaelStJules","2023","blog","LessWrong","www.lesswrong.com/posts/rPr6E2qakW9hnaxfc/utilitarianism-is-irrational-or-self-undermining",0,"","theory"],["A personal explanation of ELK concept and task.","Zeyu Qin","2023","blog","LessWrong","www.lesswrong.com/posts/AvdTogSbw2tEdMWxm/a-personal-explanation-of-elk-concept-and-task",0,"","interpretability eliciting-latent-knowledge"],["What AI could mean for animals","Max Taylor","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZNcdt7eYWW7YXALvx/what-ai-could-mean-for-animals",0,"",""],["Best project management software for research projects and labs?","PeterSlattery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vEL3aZDXTbLHAe25o/best-project-management-software-for-research-projects-and",0,"",""],["Evaluating the historical value misspecification argument","Matthew Barnett","2023","blog","LessWrong","www.lesswrong.com/posts/i5kijcjFJD6bn7dwq/evaluating-the-historical-value-misspecification-argument",0,"","evals"],["Ideation and Trajectory Modelling in Language Models","NickyP","2023","blog","LessWrong","www.lesswrong.com/posts/j9qG76qAKygPbGqZy/ideation-and-trajectory-modelling-in-language-models",0,"","interpretability"],["Pause For Thought: The AI Pause Debate (Astral Codex Ten)","David Mears","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZAxaXaakgdQK3ACqY/pause-for-thought-the-ai-pause-debate-astral-codex-ten",0,"",""],["Stampy's AI Safety Info soft launch","StevenKaas and robertskmiles","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mHNoaNvpEuzzBEEfg/stampy-s-ai-safety-info-soft-launch",0,"",""],["Stampy's AI Safety Info soft launch","steven0461 and Robert Miles","2023","blog","LessWrong","www.lesswrong.com/posts/obMiQv9K76nRZj9tE/stampy-s-ai-safety-info-soft-launch",0,"",""],["Towards Monosemanticity: Decomposing Language Models With Dictionary Learning","Zac Hatfield-Dodds","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TDqvQFks6TWutJEKu/towards-monosemanticity-decomposing-language-models-with",0,"","interpretability mechanistic-interpretability"],["AISN #23: New OpenAI Models, News from Anthropic, and Representation Engineering","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3xwSa4cE9eaxfo5mH/aisn-23-new-openai-models-news-from-anthropic-and",0,"",""],["Apply to Spring 2024 policy internships (we can help)","Elika and Vaidehi Agarwalla","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2unCFr2pnFHuNDT9z/apply-to-spring-2024-policy-internships-we-can-help",0,"","policy"],["Entanglement and intuition about words and meaning","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/DHMDxCekQbAFdyPpa/entanglement-and-intuition-about-words-and-meaning",0,"","interpretability"],["Ethical Considerations in regard to Outsourcing Labour Needs to the Global South","Nicole Mutung'a","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GzHwFz4ihnfXpPGz2/ethical-considerations-in-regard-to-outsourcing-labour-needs",0,"",""],["Fiscal sponsorship, ops support, or incubation?","Harry Luk and Peter S. Park","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zzcWFPHCuNEYCw4kJ/fiscal-sponsorship-ops-support-or-incubation",0,"",""],["Graphical tensor notation for interpretability","Jordan Taylor","2023","blog","LessWrong","www.lesswrong.com/posts/BQKKQiBmc63fwjDrj/graphical-tensor-notation-for-interpretability",0,"","interpretability mechanistic-interpretability"],["How Rethink Priorities’ Research could inform your grantmaking","kierangreig and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TT62phLw2AZWn6tDc/how-rethink-priorities-research-could-inform-your",0,"",""],["How to solve deception and still fail.","Charlie Steiner","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dWMzzd6hfimTQk8yk/how-to-solve-deception-and-still-fail",0,"","deception"],["I don’t find the lie detection results that surprising (by an author of the paper)","JanBrauner","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/noxJrzXcdz738uqMi/i-don-t-find-the-lie-detection-results-that-surprising-by-an",0,"",""],["What are some examples of AIs instantiating the 'nearest unblocked strategy problem'?","EJT","2023","blog","LessWrong","www.lesswrong.com/posts/oQ7LXJP5bzNKxomWm/what-are-some-examples-of-ais-instantiating-the-nearest",0,"","instrumental-convergence"],["Why isn't there a Charity Entrepreneurship program for AI Safety?","yanni","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LjBgFdgHGjmnwjGob/why-isn-t-there-a-charity-entrepreneurship-program-for-ai",0,"",""],["AXRP Episode 25 - Cooperative AI with Caspar Oesterheld","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CTh3JHBmEfHjE7WP5/axrp-episode-25-cooperative-ai-with-caspar-oesterheld",0,"","agents theory"],["De Dicto and De Se Reference Matters for Alignment","philgoetz","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TmnYEfiqxFtAXDaCd/de-dicto-and-de-se-reference-matters-for-alignment",0,"",""],["Early Experiments in Reward Model Interpretation Using Sparse Autoencoders","marc/er and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/QXEeis95sKrStLu2Q/early-experiments-in-reward-model-interpretation-using",0,"","interpretability mechanistic-interpretability"],["Some Quick Follow-Up Experiments to “Taken out of context: On measuring situational awareness in LLMs”","miles","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YRis8ZDstqnaW2erL/some-quick-follow-up-experiments-to-taken-out-of-context-on",0,"","situational-awareness"],["What would it mean to understand how a large language model (LLM) works? Some quick notes.","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/Be3CfAW5PMWT9nNY9/what-would-it-mean-to-understand-how-a-large-language-model",0,"","interpretability"],["Why We Use Money? - A Walrasian View","Savio Coelho","2023","blog","LessWrong","www.lesswrong.com/posts/67rThJdKAJ2C4eE4M/why-we-use-money-a-walrasian-view",0,"","theory"],["Announcing FAR Labs, an AI safety coworking space","ghabs","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9dHpEjCzBnenaXfBC/announcing-far-labs-an-ai-safety-coworking-space-1",0,"",""],["Automated Parliaments — A Solution to Decision Uncertainty and Misalignment in Language Models","Shak Ragoler","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LxvLiAhS47phswLJi/automated-parliaments-a-solution-to-decision-uncertainty-and",0,"",""],["Direction of Fit","NicholasKees","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6xKMSfK8oTpTtWKZN/direction-of-fit-1",0,"",""],["Expectations for Gemini: hopefully not a big deal","Maxime Riché","2023","blog","LessWrong","www.lesswrong.com/posts/PxELfZnvbv8jcKewp/expectations-for-gemini-hopefully-not-a-big-deal",0,"","forecasting"],["Modelling large-scale cyber attacks from advanced AI systems with Advanced Persistent Threats","Iyngkarran Kumar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bhrKwJE7Ggv7AFM7C/modelling-large-scale-cyber-attacks-from-advanced-ai-systems",0,"",""],["Observations on the funding landscape of EA and AI safety","Vilhelm Skoglund and Jona","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RueHqBuBKQBtSYkzp/observations-on-the-funding-landscape-of-ea-and-ai-safety",0,"",""],["Representation Engineering: A Top-Down Approach to AI Transparency","Andy Zou and 17 others","2023","paper","arXiv preprint","arxiv.org/abs/2310.01405",0,"","interpretability mechanistic-interpretability power-seeking monitoring"],["AI Safety Impact Markets: Your Charity Evaluator for AI Safety","Dawn Drescher","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fL7aZSq6jbDWJkzt2/ai-safety-impact-markets-your-charity-evaluator-for-ai",0,"","evals forecasting"],["Join AISafety.info's Distillation Hackathon (Oct 6-9th)","Siao Si and Stampy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vdTQnkETbECvPB3mY/join-aisafety-info-s-distillation-hackathon-oct-6-9th",0,"",""],["New Tool: the Residual Stream Viewer","AdamYedidia","2023","blog","LessWrong","www.lesswrong.com/posts/yawkDNfjR2nnzycMZ/new-tool-the-residual-stream-viewer",0,"","interpretability"],["Announcing the Winners of the 2023 Open Philanthropy AI Worldviews Contest","Jason Schukraft","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eSZuJcLGd7BacjWGi/announcing-the-winners-of-the-2023-open-philanthropy-ai",0,"",""],["Focusing your impact on short vs long TAI timelines","kuhanj","2023","blog","LessWrong","www.lesswrong.com/posts/t7Lgp77nAm4DEXPNE/focusing-your-impact-on-short-vs-long-tai-timelines",0,"","forecasting"],["How model editing could help with the alignment problem","Michael Ripa","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FNmzn33akiSesc4Ke/how-model-editing-could-help-with-the-alignment-problem",0,"",""],["Introducing Future Matters – a strategy consultancy","KyleGracey and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nb3vfv4ntM6dwQ9mx/introducing-future-matters-a-strategy-consultancy",0,"","policy"],["\"Diamondoid bacteria\" nanobots: deadly threat or dead-end? A nanotech investigation","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/g72tGduJMDhqR86Ns/diamondoid-bacteria-nanobots-deadly-threat-or-dead-end-a",0,"",""],["Anki deck for learning the main AI safety orgs, projects, and programs","Bryce Robertson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QGtYFLBtSuegHz5tP/anki-deck-for-learning-the-main-ai-safety-orgs-projects-and",0,"",""],["Steering subsystems: capabilities, agency, and alignment","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/qzu9o3sTytbC4sZkQ/steering-subsystems-capabilities-agency-and-alignment",0,"",""],["The Retroactive Funding Landscape: Innovations for Donors and Grantmakers","Dawn Drescher","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AgDrzikcHeyoHvaqd/the-retroactive-funding-landscape-innovations-for-donors-and",0,"","evals"],["Alibaba Group releases Qwen, 14B parameter LLM","nikola","2023","blog","LessWrong","www.lesswrong.com/posts/eQq4GMJTvTGNcNsnk/alibaba-group-releases-qwen-14b-parameter-llm",0,"","forecasting"],["Alignment Workshop talks","Richard_Ngo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dQ8wiAwnD37y6PkTa/alignment-workshop-talks",0,"",""],["ARC Evals: Responsible Scaling Policies","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/pnmFBjHtpfpAc6dPT/arc-evals-responsible-scaling-policies",0,"","evals governance"],["Culture and Programming Retrospective: ERA Fellowship 2023","Gideon Futerman and Nandini Shiralkar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uSKQLstvoxHKtWkrg/culture-and-programming-retrospective-era-fellowship-2023",0,"",""],["Different views of alignment have different consequences for imperfect methods","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wDvHP9KzGqPCtw3PS/different-views-of-alignment-have-different-consequences-for",0,"",""],["High-level interpretability: detecting an AI's objectives","Paul Colognese and Jozdien","2023","blog","LessWrong","www.lesswrong.com/posts/tFYGdq9ivjA3rdaS2/high-level-interpretability-detecting-an-ai-s-objectives",0,"","interpretability alignment-faking deception"],["How to Catch an AI Liar: Lie Detection in Black-Box LLMs by Asking Unrelated Questions","JanBrauner and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/khFC2a4pLPvGtXAGG/how-to-catch-an-ai-liar-lie-detection-in-black-box-llms-by",0,"",""],["Tarbell Fellowship 2024 - Applications Open (AI Journalism)","Cillian Crosson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7awJW2GPafcE4HYNf/tarbell-fellowship-2024-applications-open-ai-journalism",0,"","governance"],["Projects I would like to see (possibly at AI Safety Camp)","Linda Linsefors","2023","blog","LessWrong","www.lesswrong.com/posts/J4s5AJ3Xqc8DwAEzQ/projects-i-would-like-to-see-possibly-at-ai-safety-camp",0,"",""],["[Linkpost] Prospect Magazine - How to save humanity from extinction","jackva","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SGZrrxThGrvn3Da5u/linkpost-prospect-magazine-how-to-save-humanity-from",0,"",""],["Announcing the CNN Interpretability Competition","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HXJaKcuQqaZDJq9xz/announcing-the-cnn-interpretability-competition",0,"","interpretability"],["ARENA 2.0 - Impact Report","TheMcDouglas and Kathryn O'Rourke","2023","blog","EA Forum","forum.effectivealtruism.org/posts/C7DbrkCpSe4AdcMek/arena-2-0-impact-report",0,"",""],["ARENA 2.0 - Impact Report","TheMcDouglas","2023","blog","LessWrong","www.lesswrong.com/posts/9fbr7axHenRAL5Gkm/arena-2-0-impact-report",0,"",""],["Inside the Mind of an Aspiring Charity Entrepreneur [Follow Along] #1 - From Layoff to Co-founding in a Breathtaking Two Months","Harry Luk","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ngk6AFo5uNHB3ZKQY/inside-the-mind-of-an-aspiring-charity-entrepreneur-follow",0,"",""],["International AI Institutions: a literature review of models, examples, and proposals","MMMaas and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aztshctf3PxBnKHqF/international-ai-institutions-a-literature-review-of-models",0,"","governance policy"],["It Is Powerful, It Can't Be Aimed","Zahima","2023","blog","LessWrong","www.lesswrong.com/posts/5Yoi7JAsZm7MstbRT/it-is-powerful-it-can-t-be-aimed",0,"",""],["Let's think about...lowering the burden of proof for liability for harms associated with AI.","dEAsign","2023","blog","EA Forum","forum.effectivealtruism.org/posts/55RGoyjhc5vcEbX8o/let-s-think-about-lowering-the-burden-of-proof-for-liability",0,"",""],["News: Spanish AI image outcry + US AI workforce \"regulation\"","Ulrik Horn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9qrAhvNi27AKtyKAw/news-spanish-ai-image-outcry-us-ai-workforce-regulation",0,"","governance"],["Aim for conditional pauses","AnonResearcherMajorAILab","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BFbsqwCuuqueFRfpW/aim-for-conditional-pauses",0,"","policy"],["Amazon to invest up to $4 billion in Anthropic","Davis_Kingsley","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oAxuq5E7DsQTmxQwi/amazon-to-invest-up-to-usd4-billion-in-anthropic",0,"",""],["How to pursue a career in AI governance and coordination","Cody_Fenwick and 80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GcYTpFBfXx7WCgza6/how-to-pursue-a-career-in-ai-governance-and-coordination",0,"","governance"],["Impact stories for model internals: an exercise for interpretability researchers","jenny","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KfDh7FqwmNGExTryT/impact-stories-for-model-internals-an-exercise-for",0,"","interpretability"],["Public Opinion on AI Safety: AIMS 2023 and 2021 Summary","Janet Pauketat and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cjEaCKmRbfa5jmPop/public-opinion-on-ai-safety-aims-2023-and-2021-summary",0,"","governance policy forecasting"],["Public Opinion on AI Safety: AIMS 2023 and 2021 Summary","Jacy Reese Anthis and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/4v59asmKZxumHmYyz/public-opinion-on-ai-safety-aims-2023-and-2021-summary",0,"","governance"],["Understanding strategic deception and deceptive alignment","Marius Hobbhahn and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fsbcq9z7korjBTP8Z/understanding-strategic-deception-and-deceptive-alignment",0,"","alignment-faking deception"],["Understanding strategic deception and deceptive alignment","Marius Hobbhahn and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/fsbcq9z7korjBTP8Z/understanding-strategic-deception-and-deceptive-alignment",0,"","alignment-faking deception"],["Welcome to Apply: The 2024 Vitalik Buterin Fellowships in AI Existential Safety by FLI!","Zhijing Jin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EjGowxHhRifb2r8tE/welcome-to-apply-the-2024-vitalik-buterin-fellowships-in-ai",0,"",""],["What causes a decision theory to be used?","Dagon","2023","blog","LessWrong","www.lesswrong.com/posts/Gj4GPdLntgJX9kodL/what-causes-a-decision-theory-to-be-used",0,"","theory"],["What is wrong with this \"utility switch button problem\" approach?","Donald Hobson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/P8caCHGJdm2GcniAp/what-is-wrong-with-this-utility-switch-button-problem",0,"",""],["“X distracts from Y” as a thinly-disguised fight over group status / politics","Steven Byrnes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NfgMAS67nKTGzmQMB/x-distracts-from-y-as-a-thinly-disguised-fight-over-group",0,"",""],["Five neglected work areas that could reduce AI risk","Aaron_Scher and Charlotte","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2RCAkouYpiKyn4AbA/five-neglected-work-areas-that-could-reduce-ai-risk",0,"","governance"],["Five neglected work areas that could reduce AI risk","CharlotteS and Aaron_Scher","2023","blog","LessWrong","www.lesswrong.com/posts/fEAGPyHR9GaK2cwRq/five-neglected-work-areas-that-could-reduce-ai-risk",0,"","governance"],["Unions for AI safety?","dEAsign","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GNfWT8Xqh89wRaaSg/unions-for-ai-safety",0,"",""],["\"We can Prevent AI Disaster Like We Prevented Nuclear Catastrophe\"","Peter","2023","blog","EA Forum","forum.effectivealtruism.org/posts/upZJFAFPeJkxFtb8i/we-can-prevent-ai-disaster-like-we-prevented-nuclear",0,"","policy"],["I designed an AI safety course (for a philosophy department)","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gY9TjxNgSkgsMRDnL/i-designed-an-ai-safety-course-for-a-philosophy-department",0,"",""],["I designed an AI safety course (for a philosophy department)","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/t3ngnd6Wvo4qeY5FA/i-designed-an-ai-safety-course-for-a-philosophy-department",0,"",""],["It’s not obvious that getting dangerous AI later is better","Aaron_Scher","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bLWG7onTMKzdozez8/it-s-not-obvious-that-getting-dangerous-ai-later-is-better",0,"","governance"],["Taking features out of superposition with sparse autoencoders more quickly with informed initialization","Pierre Peigné","2023","blog","LessWrong","www.lesswrong.com/posts/YJpMgi7HJuHwXTkjk/taking-features-out-of-superposition-with-sparse",0,"","interpretability mechanistic-interpretability"],["Evidence to prioritize or working on AI as the most impactful thing?","Vaipan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4NPDqtb8oWHBJMJTN/evidence-to-prioritize-or-working-on-ai-as-the-most-1",0,"",""],["How could a moratorium fail?","Davidmanheim","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fwdjMtJLpkyJ2Gice/how-could-a-moratorium-fail",0,"",""],["Intro to AI risk for AI grad students?","tae","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xJYRiy8Jjy2Tk2qHr/intro-to-ai-risk-for-ai-grad-students",0,"",""],["Invitation to review: my draft submission to the UN Tech Envoy's Multistakeholder Advisory Body on Artificial Intelligence​","MattThinks","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rBEjfArWxaWEf9yNE/invitation-to-review-my-draft-submission-to-the-un-tech",0,"","governance policy"],["Let's talk about Impostor syndrome in AI safety","Igor Ivanov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rwW8GKAfuagKgG7AQ/let-s-talk-about-impostor-syndrome-in-ai-safety",0,"",""],["We are not alone: many communities want to stop Big Tech from scaling unsafe AI","Remmelt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/q8jxedwSKBdWA3nH7/we-are-not-alone-many-communities-want-to-stop-big-tech-from",0,"",""],["AI is centralizing by default; let's not make it worse","Quintin Pope","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zd5inbT4kYKivincm/ai-is-centralizing-by-default-let-s-not-make-it-worse",0,"",""],["Is there much need for frontend engineers in AI alignment?","Michael G","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yWHhb5MBBD7gb46Ch/is-there-much-need-for-frontend-engineers-in-ai-alignment",0,"",""],["Sparse Autoencoders Find Highly Interpretable Directions in Language Models","Logan Riggs and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qryk6FqjtZk9FHHJR/sparse-autoencoders-find-highly-interpretable-directions-in",0,"","interpretability mechanistic-interpretability"],["Sparse Autoencoders Find Highly Interpretable Directions in Language Models","Logan Riggs and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/Qryk6FqjtZk9FHHJR/sparse-autoencoders-find-highly-interpretable-directions-in",0,"","interpretability mechanistic-interpretability"],["Sparse Autoencoders: Future Work","Logan Riggs and Aidan Ewart","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CkFBMG6A9ytkiXBDM/sparse-autoencoders-future-work",0,"","mechanistic-interpretability"],["The “technology\" bucket error","Holly_Elmore","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TPDtmSnJbGZFDZTfs/the-technology-bucket-error",0,"",""],["There should be more AI safety orgs","Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MhudbfBNQcMxBBvj8/there-should-be-more-ai-safety-orgs",0,"",""],["Careless talk on US-China AI competition? (and criticism of CAIS coverage)","Oliver Sourbut","2023","blog","LessWrong","www.lesswrong.com/posts/uRyKkyYstxZkCNcoP/careless-talk-on-us-china-ai-competition-and-criticism-of",0,"","governance"],["Existential Cybersecurity Risks & AI (A Research Agenda)","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KHmoNx3zpCAaiHxTW/existential-cybersecurity-risks-and-ai-a-research-agenda",0,"",""],["Interpretability Externalities Case Study - Hungry Hungry Hippos","Magdalena Wache","2023","blog","LessWrong","www.lesswrong.com/posts/75uJN3qqzyxWoknN7/interpretability-externalities-case-study-hungry-hungry",0,"","interpretability"],["The Case for AI Safety Advocacy to the Public","Holly_Elmore","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Y4SaFM5LfsZzbnymu/the-case-for-ai-safety-advocacy-to-the-public",0,"",""],["[Link post] Michael Nielsen's \"Notes on Existential Risk from Artificial Superintelligence\"","Joel Becker","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5NcCWNC3yWdqeaEdH/link-post-michael-nielsen-s-notes-on-existential-risk-from",0,"",""],["AISN #22: The Landscape of US AI Legislation - Hearings, Frameworks, Bills, and Laws","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MEN9YMyqBJ9AodZri/aisn-22-the-landscape-of-us-ai-legislation-hearings",0,"","policy"],["Anthropic's Responsible Scaling Policy & Long-Term Benefit Trust","Zac Hatfield-Dodds","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6tjHf5ykvFqaNCErH/anthropic-s-responsible-scaling-policy-and-long-term-benefit",0,"","policy"],["Anthropic's Responsible Scaling Policy & Long-Term Benefit Trust","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bGzwWYfXgKqdurdmb/anthropic-s-responsible-scaling-policy-and-long-term-benefit",0,"","policy"],["Formalizing «Boundaries» with Markov blankets + Criticism of this approach","Chipmonk","2023","blog","LessWrong","www.lesswrong.com/posts/z4o4iAFgnmaBmksN2/formalizing-boundaries-with-markov-blankets-criticism-of",0,"",""],["Protest against Meta's irreversible proliferation (Sept 29, San Francisco)","Holly_Elmore","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iitD7ia96CYkocLTd/protest-against-meta-s-irreversible-proliferation-sept-29",0,"",""],["The possibility of an indefinite AI pause","Matthew_Barnett","2023","blog","EA Forum","forum.effectivealtruism.org/posts/k6K3iktCLCTHRMJsY/the-possibility-of-an-indefinite-ai-pause",0,"",""],["Comments on Manheim's \"What's in a Pause?\"","RobBensinger","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fSeDA7B7Hve5LeaWq/comments-on-manheim-s-what-s-in-a-pause",0,"",""],["Knowledge Database 1: The structure and the method of building","iwis","2023","blog","LessWrong","www.lesswrong.com/posts/FqdT8vpwiDKFYQHFR/knowledge-database-1-the-structure-and-the-method-of-1",0,"",""],["Knowledge Database 2: Shopping advisor and other uses of knowledge base about products","iwis","2023","blog","LessWrong","www.lesswrong.com/posts/hNFQSGfvfPgHvCryT/knowledge-database-2-shopping-advisor-and-other-uses-of-1",0,"",""],["Relationship between EA Community and AI safety","Tom Barnes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/opCxiPwxFcaaayyMB/relationship-between-ea-community-and-ai-safety",0,"",""],["Stuart J. Russell on \"should we press pause on AI?\"","Kaleem","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KYGuGAyZQwAgecQg6/stuart-j-russell-on-should-we-press-pause-on-ai",0,"",""],["Technical AI Safety Research Landscape [Slides]","Magdalena Wache","2023","blog","LessWrong","www.lesswrong.com/posts/x2n7mBLryDXuLwGhx/technical-ai-safety-research-landscape-slides",0,"",""],["The omnizoid - Heighn FDT Debate #5","Heighn","2023","blog","LessWrong","www.lesswrong.com/posts/tKRGW7fEqCgStCgEd/the-omnizoid-heighn-fdt-debate-5",0,"","theory"],["US public opinion on AI, September 2023","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/G7bWx2XaBxrBEKGgh/us-public-opinion-on-ai-september-2023",0,"",""],["Where might I direct promising-to-me researchers to apply for alignment jobs/grants?","abramdemski","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kxFwEKRy3vjCWDNm4/where-might-i-direct-promising-to-me-researchers-to-apply",0,"",""],["Catalyst books","Catnee","2023","blog","LessWrong","www.lesswrong.com/posts/nkWZAEopxvwRTAP5D/catalyst-books",0,"",""],["How to talk about reasons why AGI might not be near?","Kaj_Sotala","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8XAxbsdtLmMaf5zta/how-to-talk-about-reasons-why-agi-might-not-be-near",0,"",""],["How to think about slowing AI","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fZmQ6WQ6MQPa5q39R/how-to-think-about-slowing-ai",0,"",""],["Microdooms averted by working on AI Safety","Nikola","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DBDpnAhxvRWmfmtfv/microdooms-averted-by-working-on-ai-safety",0,"","forecasting"],["Microdooms averted by working on AI Safety","nikola","2023","blog","LessWrong","www.lesswrong.com/posts/mTtxJKN3Ew8CAEHGr/microdooms-averted-by-working-on-ai-safety",0,"",""],["Reflexive decision theory is an unsolved problem","Richard_Kennaway","2023","blog","LessWrong","www.lesswrong.com/posts/JHgSgwjnBpcirfq9m/reflexive-decision-theory-is-an-unsolved-problem",0,"","theory"],["Telopheme, telophore, and telotect","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/p7mMJvwDbuvo4K7NE/telopheme-telophore-and-telotect",0,"",""],["AI Pause Will Likely Backfire","Nora Belrose","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JYEAL8g7ArqGoTaX6/ai-pause-will-likely-backfire",0,"",""],["Policy ideas for mitigating AI risk","Thomas Larsen","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DG6bf5YW3jxLRD7KN/policy-ideas-for-mitigating-ai-risk",0,"","policy"],["What's in a Pause?","Davidmanheim","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3hSEQnEN2D3SSzHWn/what-s-in-a-pause-3",0,"",""],["A conversation with Pi, a conversational AI.","Spiritus Dei","2023","blog","LessWrong","www.lesswrong.com/posts/pzvHZsKyJZks89Pao/a-conversation-with-pi-a-conversational-ai",0,"","forecasting"],["Cruxes for overhang","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/59dKN8XQGx952irWg/cruxes-for-overhang-1",0,"","forecasting"],["Destroying the fabric of the universe as an instrumental goal.","AI-doom","2023","blog","LessWrong","www.lesswrong.com/posts/3wg3YBmkukWzecyR9/destroying-the-fabric-of-the-universe-as-an-instrumental",0,"","instrumental-convergence"],["Instrumental Convergence Bounty","Logan Zoellner","2023","blog","LessWrong","www.lesswrong.com/posts/ym4BAovbgLAaXsf79/instrumental-convergence-bounty",0,"","instrumental-convergence"],["The state of AI in different countries — an overview","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Lb2TjSsjpqA8rQ7dP/the-state-of-ai-in-different-countries-an-overview",0,"","governance"],["Uncovering Latent Human Wellbeing in LLM Embeddings","ChengCheng and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/BDTZBPunnvffCfKff/uncovering-latent-human-wellbeing-in-llm-embeddings",0,"","interpretability eliciting-latent-knowledge"],["AI-Risk in the State of the European Union Address","Sam Bogerd","2023","blog","EA Forum","forum.effectivealtruism.org/posts/twf3ByYGZGAupAKAB/ai-risk-in-the-state-of-the-european-union-address",0,"","policy"],["Applications for EU Tech Policy Fellowship 2024 now open","Jan-Willem and Training for Good","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qrkiKXAHy6z7yABGv/applications-for-eu-tech-policy-fellowship-2024-now-open",0,"","governance policy"],["Apply to lead a project during the next virtual AI Safety Camp","Linda Linsefors and Remmelt","2023","blog","LessWrong","www.lesswrong.com/posts/mw8X3wCdcHipdTicv/apply-to-lead-a-project-during-the-next-virtual-ai-safety",0,"",""],["Is AI Safety dropping the ball on privacy?","markov","2023","blog","LessWrong","www.lesswrong.com/posts/WCevxhGtmnPhWH3ah/is-ai-safety-dropping-the-ball-on-privacy-1",0,"",""],["MLSN: #10 Adversarial Attacks Against Language and Vision Models, Improving LLM Honesty, and Tracing the Influence of LLM Training Data","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zpDr8ZmNxkEQqTkNo/mlsn-10-adversarial-attacks-against-language-and-vision",0,"","training-data"],["UDT shows that decision theory is more puzzling than ever","Wei Dai","2023","blog","LessWrong","www.lesswrong.com/posts/wXbSAKu2AcohaK2Gt/udt-shows-that-decision-theory-is-more-puzzling-than-ever",0,"","theory"],["Who should we interview for The 80,000 Hours Podcast?","Luisa_Rodriguez and Robert_Wiblin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RvafKqEYndLrnrGjm/who-should-we-interview-for-the-80-000-hours-podcast",0,"",""],["Automatically finding feature vectors in the OV circuits of Transformers without using probing","Jacob Dunefsky","2023","blog","LessWrong","www.lesswrong.com/posts/jDfjqu2qJLcPco9cf/automatically-finding-feature-vectors-in-the-ov-circuits-of",0,"","interpretability mechanistic-interpretability"],["How useful is Corrigibility?","martinkunev","2023","blog","LessWrong","www.lesswrong.com/posts/Py3vqPp9uSqQJHFuy/how-useful-is-corrigibility",0,"",""],["Theory: “WAW might be of higher impact than x-risk prevention based on utilitarianism”","Jens Aslaug","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TQdNRM9gofsN9thYv/theory-waw-might-be-of-higher-impact-than-x-risk-prevention",0,"","forecasting"],["Focus on the Hardest Part First","Johannes C. Mayer","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Eav2BizSejDcztFC8/focus-on-the-hardest-part-first",0,"",""],["How should technical AI researchers best transition into AI governance and policy?","Gabriel Mukobi","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WDsxB6n4dhxmwbQex/how-should-technical-ai-researchers-best-transition-into-ai",0,"","governance policy"],["How teams went about their research at AI Safety Camp edition 8","Remmelt and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/EsBDbtnCizxDLsDT3/how-teams-went-about-their-research-at-ai-safety-camp-1",0,"",""],["Panel discussion on AI consciousness with Rob Long and Jeff Sebo","Aaron Bergman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pNhc3jensyBY4Hz6u/panel-discussion-on-ai-consciousness-with-rob-long-and-jeff",0,"",""],["Possible Divergence in AGI Risk Tolerance between Selfish and Altruistic agents","Brad West","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ggSXcuMzRaowDbKTz/possible-divergence-in-agi-risk-tolerance-between-selfish",0,"","agents"],["US presidents discuss AI alignment agendas","TurnTrout and Garrett Baker","2023","blog","LessWrong","www.lesswrong.com/posts/7M2iHPLaNzPNXHuMv/us-presidents-discuss-ai-alignment-agendas",0,"",""],["A case study of regulation done well? Canadian biorisk regulations","rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cKbehBhq7NxTq3pck/a-case-study-of-regulation-done-well-canadian-biorisk",0,"","governance"],["Debate series: should we push for a pause on the development of AI?","Ben_West","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6SvZPHAvhT5dtqefF/debate-series-should-we-push-for-a-pause-on-the-development",0,"",""],["Explained Simply: Quantilizers","brook","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QhzJFpQPa9qxfAmXp/explained-simply-quantilizers",0,"",""],["Explaining grokking through circuit efficiency","Vikrant Varma and Rohin Shah","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JK2QGfNGLjuFnrEvz/explaining-grokking-through-circuit-efficiency",0,"","interpretability mechanistic-interpretability"],["Recreating the caring drive","Catnee","2023","blog","LessWrong","www.lesswrong.com/posts/JjqZexMgvarBFMKPs/recreating-the-caring-drive",0,"",""],["How long will reaching a Risk Awareness Moment and CHARTS agreement take?","Yadav","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hybGfBnkrtL9E3EcS/how-long-will-reaching-a-risk-awareness-moment-and-charts",0,"","governance policy"],["What I would do if I wasn’t at ARC Evals","Lawrence Chan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zcHdehWJzDpfxJpmf/what-i-would-do-if-i-wasn-t-at-arc-evals",0,"","evals"],["What term to use for AI in different policy contexts?","oeg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9Y5YzNDMdYYg6hjwD/what-term-to-use-for-ai-in-different-policy-contexts",0,"","governance policy"],["What's in your list of important technical projects/experiments to run for AI alignment?","watermark","2023","blog","LessWrong","www.lesswrong.com/posts/ZNXiKT3dNvLzj7d6m/what-s-in-your-list-of-important-technical-projects",0,"",""],["AISN #21: Google DeepMind’s GPT-4 Competitor, Military Investments in Autonomous Drones, The UK AI Safety Summit, and Case Studies in AI Policy","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9eQFPiNmH2s5ZyNEu/aisn-21-google-deepmind-s-gpt-4-competitor-military",0,"","policy"],["AISN #21: Google DeepMind’s GPT-4 Competitor, Military Investments in Autonomous Drones, The UK AI Safety Summit, and Case Studies in AI Policy","aogara and Dan H","2023","blog","LessWrong","www.lesswrong.com/posts/jkEEHfQkwvbLkzpiF/aisn-21-google-deepmind-s-gpt-4-competitor-military",0,"","policy"],["Benchmarks for Detecting Measurement Tampering [Redwood Research]","ryan_greenblatt and Fabien Roger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/inALbAqdx63KTaGgs/benchmarks-for-detecting-measurement-tampering-redwood",0,"","benchmarks robustness"],["Decision theory is not policy theory is not agent theory","Cole Wyeth","2023","blog","LessWrong","www.lesswrong.com/posts/MwetLcBPvshg9ePZB/decision-theory-is-not-policy-theory-is-not-agent-theory",0,"","agents policy theory"],["Strongest real-world examples supporting AI risk claims?","rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GDdvdhbGfCehnoJzY/strongest-real-world-examples-supporting-ai-risk-claims",0,"",""],["The Evolutionary Pathway from Biological to Digital Intelligence: A Cosmic Perspective","George360","2023","blog","LessWrong","www.lesswrong.com/posts/FXWwsTWAjTwCtZmQj/the-evolutionary-pathway-from-biological-to-digital",0,"",""],["What I would do if I wasn’t at ARC Evals","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6FkWnktH3mjMAxdRT/what-i-would-do-if-i-wasn-t-at-arc-evals",0,"","evals"],["[CFP] NeurIPS workshop: AI meets Moral Philosophy and Moral Psychology","jaredlcm","2023","blog","EA Forum","forum.effectivealtruism.org/posts/B3NyGg24gtdKETnXw/cfp-neurips-workshop-ai-meets-moral-philosophy-and-moral",0,"",""],["Against the Open Source / Closed Source Dichotomy: Regulated Source as a Model for Responsible AI Development","alexherwix","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GBAzW4pZ5JgJqGMJg/against-the-open-source-closed-source-dichotomy-regulated",0,"","policy"],["Data Poisoning for Dummies (No Code, No Math)","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bYm63mL6NioCMq66w/data-poisoning-for-dummies-no-code-no-math",0,"","training-data"],["Getting Washington and Silicon Valley to tame AI (Mustafa Suleyman on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/d4mr2GDftfsh8BDpq/getting-washington-and-silicon-valley-to-tame-ai-mustafa",0,"",""],["Hertford, Sourbut (rationality lessons from University Challenge)","Oliver Sourbut","2023","blog","LessWrong","www.lesswrong.com/posts/xkusvgfxD8MbDtxin/hertford-sourbut-rationality-lessons-from-university",0,"","theory"],["No. Impending AGI doesn't make everything else unimportant.","Igor Ivanov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/prvzqAxbRtzAcorq6/no-impending-agi-doesn-t-make-everything-else-unimportant",0,"",""],["Notes on nukes, IR, and AI from \"Arsenals of Folly\" (and other books)","tlevin","2023","blog","LessWrong","www.lesswrong.com/posts/njEWACBHhfppg6KYS/notes-on-nukes-ir-and-ai-from-arsenals-of-folly-and-other",0,"","governance"],["Paper: On measuring situational awareness in LLMs","Owain_Evans and 7 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mLfPHv4QjmeQrsSva/paper-on-measuring-situational-awareness-in-llms",0,"","alignment-faking deception situational-awareness scaling-laws"],["Transformative AI and Compute - Reading List","Frederik Berg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3Rz4TGR9T8JELGGG9/transformative-ai-and-compute-reading-list",0,"","governance compute-governance forecasting"],["[Linkpost] Beware the Squirrel by Verity Harding","Arden","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SkEgBb5KqNRWHcksa/linkpost-beware-the-squirrel-by-verity-harding",0,"",""],["Fundamental question: What determines a mind's effects?","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NqsNYsyoA2YSbb3py/fundamental-question-what-determines-a-mind-s-effects",0,"",""],["Series of absurd upgrades in nature's great search","lukehmiles","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/o2BkyQLrscbLbQSJn/series-of-absurd-upgrades-in-nature-s-great-search",0,"",""],["Is AI like disk drives?","Tanae","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kJWqg4JjGcJCF5gyj/is-ai-like-disk-drives",0,"","governance"],["PIBBSS Summer Symposium 2023","Nora_Ammann and DusanDNesic","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/86KqAYdjrW7niXydq/pibbss-summer-symposium-2023",0,"",""],["Rational Agents Cooperate in the Prisoner's Dilemma","Isaac King","2023","blog","LessWrong","www.lesswrong.com/posts/QX98rCSXPkPSMriYi/rational-agents-cooperate-in-the-prisoner-s-dilemma",0,"","agents theory"],["AI pause/governance advocacy might be net-negative, especially without focus on explaining the x-risk","Samin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/y7pCAoghcNKhhufCS/ai-pause-governance-advocacy-might-be-net-negative",0,"","governance"],["Meta Questions about Metaphilosophy","Wei Dai","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fJqP9WcnHXBRBeiBg/meta-questions-about-metaphilosophy",0,"",""],["[Linkpost] Michael Nielsen remarks on 'Oppenheimer'","Tom Barnes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ph6wvA2EtQ7pG3yvG/linkpost-michael-nielsen-remarks-on-oppenheimer",0,"",""],["Alignment & Capabilities: What's the difference?","John G. Halstead","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sXJkaQFFYodhEXNvr/alignment-and-capabilities-what-s-the-difference",0,"",""],["Should some people start working to influence the people who are most likely to shape the values of the first AGIs, so that they take into account the interests of wild and farmed animals and sentient digital minds?","Keyvan Mostafavi","2023","blog","EA Forum","forum.effectivealtruism.org/posts/svaHSiPFykYs9tYet/should-some-people-start-working-to-influence-the-people-who",0,"",""],["\"Wanting\" and \"liking\"","Mateusz Bagiński","2023","blog","LessWrong","www.lesswrong.com/posts/opJxxfrN33xQx3eXu/wanting-and-liking",0,"",""],["Agency Foundations Challenge: September 8th-24th, $10k Prizes","Catalin M and Esben Kran","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DLrhnzDeqSywhrNew/agency-foundations-challenge-september-8th-24th-usd10k",0,"",""],["An adversarial example for Direct Logit Attribution: memory management in gelu-4l","Can Rager and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/2PucFqdRyEvaHb4Hn/an-adversarial-example-for-direct-logit-attribution-memory",0,"","interpretability mechanistic-interpretability robustness"],["Invulnerable Incomplete Preferences: A Formal Statement","Sami Petersen","2023","blog","LessWrong","www.lesswrong.com/posts/sHGxvJrBag7nhTQvb/invulnerable-incomplete-preferences-a-formal-statement-1",0,"",""],["Report on Frontier Model Training","YafahEdelman","2023","blog","LessWrong","www.lesswrong.com/posts/nXcHe7t4rqHMjhzau/report-on-frontier-model-training",0,"","governance forecasting"],["Responses to apparent rationalist confusions about game / decision theory","Anthony DiGiovanni","2023","blog","LessWrong","www.lesswrong.com/posts/895Qmhyud2PjDhte6/responses-to-apparent-rationalist-confusions-about-game",0,"","theory"],["Updates from Campaign for AI Safety","Jolyn Khoo and Nik Samoylov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/73mcXk9gEq6GpgpFG/updates-from-campaign-for-ai-safety-4",0,"",""],["AI Deception: A Survey of Examples, Risks, and Potential Solutions","Simon Goldstein and Peter S. Park","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YgAKhkBdgeTCn6P53/ai-deception-a-survey-of-examples-risks-and-potential",0,"","deception"],["AISN #20: LLM Proliferation, AI Deception, and Continuing Drivers of AI Capabilities","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Hg4dQqxyFpmkoYKeg/aisn-20-llm-proliferation-ai-deception-and-continuing",0,"","deception"],["An Interpretability Illusion for Activation Patching of Arbitrary Subspaces","Georg Lange and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RFtkRXHebkwxygDe2/an-interpretability-illusion-for-activation-patching-of",0,"","interpretability mechanistic-interpretability"],["An OV-Coherent Toy Model of Attention Head Superposition","LaurenGreenspan and keith_wynroe","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cqRGZisKbpSjgaJbc/an-ov-coherent-toy-model-of-attention-head-superposition-1",0,"","interpretability mechanistic-interpretability"],["Anyone want to debate publicly about FDT?","omnizoid","2023","blog","LessWrong","www.lesswrong.com/posts/Wvyiycf6dHeknJRRA/anyone-want-to-debate-publicly-about-fdt",0,"","theory"],["Apply to a small iteration of MLAB to be run in Oxford","Rio P and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AhKpFhL4gKErf7bo3/apply-to-a-small-iteration-of-mlab-to-be-run-in-oxford",0,"",""],["Barriers to Mechanistic Interpretability for AGI Safety","Connor Leahy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KRDo2afKJtD7bzSM8/barriers-to-mechanistic-interpretability-for-agi-safety",0,"","interpretability mechanistic-interpretability"],["Democratic Fine-Tuning","Joe Edelman","2023","blog","LessWrong","www.lesswrong.com/posts/ncb2ycEB3ymNqzs93/democratic-fine-tuning",0,"",""],["Impact Academy is hiring an AI Governance Lead - more information, upcoming Q&A and $500 bounty","Lowe and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EB9wCLvRjJiQt7DyS/impact-academy-is-hiring-an-ai-governance-lead-more",0,"","governance policy"],["Incentives affecting alignment-researcher encouragement","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/wBgjQKNfJnPMjKpFa/incentives-affecting-alignment-researcher-encouragement",0,"",""],["Language models surprised us","Ajeya","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gYoB8vZcPGcL3cAKH/language-models-surprised-us",0,"",""],["Newcomb Variant","lsusr","2023","blog","LessWrong","www.lesswrong.com/posts/jzfLjhgMrCb5sE2Go/newcomb-variant",0,"","theory"],["OpenAI API base models are not sycophantic, at any size","nostalgebraist","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3ou8DayvDXxufkjHD/openai-api-base-models-are-not-sycophantic-at-any-size",0,"","sycophancy"],["Paper Walkthrough: Automated Circuit Discovery with Arthur Conmy","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FzqKXpTDaouMF6Chj/paper-walkthrough-automated-circuit-discovery-with-arthur",0,"","interpretability mechanistic-interpretability"],["AI Deception: A Survey of Examples, Risks, and Potential Solutions","Peter S. Park","2023","paper","arXiv preprint","arxiv.org/abs/2308.14752",0,"","deception policy"],["Information warfare historically revolved around human conduits","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/c5oyHuHaw4AcWy4tf/information-warfare-historically-revolved-around-human",0,"","governance forecasting"],["Introducing the Center for AI Policy (& we're hiring!)","Thomas Larsen","2023","blog","LessWrong","www.lesswrong.com/posts/unwRBRQivd2LYRfuP/introducing-the-center-for-ai-policy-and-we-re-hiring",0,"","governance policy"],["Navigating the Future: A Guide on How to Stay Safe with AI | Emmanuel Katto Uganda","emmanuelkatto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XxrvEytsHQAM38MBt/navigating-the-future-a-guide-on-how-to-stay-safe-with-ai-or",0,"",""],["Paradigms and Theory Choice in AI: Adaptivity, Economy and Control","particlemania","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TEDT4SJDfBwXezfgC/paradigms-and-theory-choice-in-ai-adaptivity-economy-and",0,"",""],["Rethink Priorities is looking for a (Co-)Founder for a New Project: Field Building in Universities for AI Policy Careers in the US","KevinN","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3J8aBk8wc668CJnbb/rethink-priorities-is-looking-for-a-co-founder-for-a-new",0,"","governance policy"],["AI pause/governance advocacy might be net-negative, especially without focus on explaining the x-risk","Mikhail Samin","2023","blog","LessWrong","www.lesswrong.com/posts/jiYLFomPPePy85eN8/ai-pause-governance-advocacy-might-be-net-negative",0,"","governance"],["Apply to a small iteration of MLAB to be run in Oxford","RP and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/k5anbk2pBZPFkrCqh/apply-to-a-small-iteration-of-mlab-to-be-run-in-oxford",0,"",""],["The Game of Dominance","Karl von Wendt","2023","blog","LessWrong","www.lesswrong.com/posts/NCDakH4nZrS9qeuL6/the-game-of-dominance",0,"","instrumental-convergence power-seeking"],["A list of core AI safety problems and how I hope to solve them","davidad","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mnoc3cKY3gXMrTybs/a-list-of-core-ai-safety-problems-and-how-i-hope-to-solve",0,"",""],["EA is underestimating intelligence agencies and this is dangerous","trevor1","2023","blog","EA Forum","forum.effectivealtruism.org/posts/L8kEmQgghxS9LXF3H/ea-is-underestimating-intelligence-agencies-and-this-is",0,"","policy forecasting"],["Mesa-Optimization: Explain it like I'm 10 Edition","brook","2023","blog","LessWrong","www.lesswrong.com/posts/fLAvmWHmpJEiw8KEp/mesa-optimization-explain-it-like-i-m-10-edition",0,"","alignment-faking deception"],["Ramble on STUFF: intelligence, simulation, AI, doom, default mode, the usual","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/i4LjHb6enWiErXdx2/ramble-on-stuff-intelligence-simulation-ai-doom-default-mode",0,"",""],["Red-teaming language models via activation engineering","Nina Rimsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iHmsJdxgMEWmAfNne/red-teaming-language-models-via-activation-engineering",0,"","red-teaming"],["A Model-based Approach to AI Existential Risk","Sammy Martin and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sGkRDrpphsu6Jhega/a-model-based-approach-to-ai-existential-risk",0,"","forecasting"],["A model-based approach to AI Existential Risk","SammyDMartin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ExpBagkng6QSqcN8d/a-model-based-approach-to-ai-existential-risk",0,"","forecasting"],["On whether AI will soon cause job loss, lower incomes, and higher inequality — or the opposite (Michael Webb on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/biyLvsheYcD6n8bqw/on-whether-ai-will-soon-cause-job-loss-lower-incomes-and",0,"",""],["What AI Posts Do You Want Distilled?","brook","2023","blog","EA Forum","forum.effectivealtruism.org/posts/m6BR4pmgXjoKJBfmt/what-ai-posts-do-you-want-distilled",0,"",""],["[Crosspost] AI Regulation May Be More Important Than AI Alignment For Existential Safety","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vZWkDkvc3zhdLaPpd/crosspost-ai-regulation-may-be-more-important-than-ai",0,"","governance policy"],["AI Regulation May Be More Important Than AI Alignment For Existential Safety","otto.barten","2023","blog","LessWrong","www.lesswrong.com/posts/2cxNvPtMrjwaJrtoR/ai-regulation-may-be-more-important-than-ai-alignment-for",0,"","governance"],["AI Safety Bounties","PatrickL","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3rf99yiGhjDdBDeCJ/ai-safety-bounties",0,"",""],["Assessment of intelligence agency functionality is difficult yet important","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/foM8SA3ftY94MGMq9/assessment-of-intelligence-agency-functionality-is-difficult",0,"","governance forecasting"],["Enhancing Corrigibility in AI Systems through Robust Feedback Loops","Justausername","2023","blog","LessWrong","www.lesswrong.com/posts/4mGDZurjv6j8AWhNe/enhancing-corrigibility-in-ai-systems-through-robust",0,"","interpretability"],["Health, morality, and goal alignment of systems, agents, and organs","FalseCogs","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6y4nS9A6WYiaFS9Kp/health-morality-and-goal-alignment-of-systems-agents-and",0,"","agents"],["Would it be useful to collect the contexts, where various LLMs think the same?","Martin Vlach","2023","blog","LessWrong","www.lesswrong.com/posts/AGPgMBp6eN95uxJyc/would-it-be-useful-to-collect-the-contexts-where-various",0,"","interpretability"],["Do agents with (mutually known) identical utility functions but irreconcilable knowledge sometimes fight?","mako yass","2023","blog","LessWrong","www.lesswrong.com/posts/FTdtHHBPDdzk4pJz2/do-agents-with-mutually-known-identical-utility-functions",0,"","agents theory"],["Implications of evidential cooperation in large worlds","Lukas Finnveden","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EeXSjvyQge5FZPeuL/implications-of-evidential-cooperation-in-large-worlds",0,"",""],["The Ethical Basilisk Thought Experiment","Kyrtin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vaARSdTS2X73jAMZ9/the-ethical-basilisk-thought-experiment",0,"",""],["Why Is No One Trying To Align Profit Incentives With Alignment Research?","Prometheus","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AYGNbYeB7bHjwidiz/why-is-no-one-trying-to-align-profit-incentives-with",0,"",""],["An argument for accelerating international AI governance research (part 2)","MattThinks","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yjvhEqBshzxMjKL9g/an-argument-for-accelerating-international-ai-governance-1",0,"","governance policy"],["Why does an AI have to have specified goals?","Luke Eure","2023","blog","EA Forum","forum.effectivealtruism.org/posts/T4EfQm9YzYdWd6Xyq/why-does-an-ai-have-to-have-specified-goals",0,"",""],["Causality and a Cost Semantics for Neural Networks","scottviteri","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zkfmhWQXsZweijmzi/causality-and-a-cost-semantics-for-neural-networks",0,"","interpretability"],["Ideas for improving epistemics in AI safety outreach","mic","2023","blog","LessWrong","www.lesswrong.com/posts/SDpaZ7MdH5yRnobrZ/ideas-for-improving-epistemics-in-ai-safety-outreach",0,"",""],["Import AI 337: Why I am confused about AI; penguin dataset; and defending networks via RL with CYBERFORCE","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-337-why-i-am-confused-about",0,"",""],["Large Language Models will be Great for Censorship","Ethan Edwards","2023","blog","LessWrong","www.lesswrong.com/posts/oqvsR2LmHWamyKDcj/large-language-models-will-be-great-for-censorship",0,"","governance"],["Self-shutdown AI","jan betley","2023","blog","LessWrong","www.lesswrong.com/posts/piAnXc2a4k5bFsKjL/self-shutdown-ai",0,"",""],["Call for Papers on Global AI Governance from the UN","Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AKnBQboyyKz9QdD4T/call-for-papers-on-global-ai-governance-from-the-un",0,"","governance policy"],["Jan Kulveit's Corrigibility Thoughts Distilled","brook","2023","blog","LessWrong","www.lesswrong.com/posts/Lg4voqq4vTXiCJNQP/jan-kulveit-s-corrigibility-thoughts-distilled",0,"",""],["Longtermism Fund: August 2023 Grants Report","Michael Townsend and Giving What We Can","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4yZzSziCLkdzsYHt6/longtermism-fund-august-2023-grants-report",0,"",""],["Memetic Judo #3: The Intelligence of Stochastic Parrots v.2","Max TK","2023","blog","LessWrong","www.lesswrong.com/posts/7aHCZbofofA5JeKgb/memetic-judo-3-the-intelligence-of-stochastic-parrots-v-2",0,"","interpretability"],["XPT forecasts on (some) Direct Approach model inputs","Forecasting Research Institute and rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/saAXc8zsFgZuxFM6L/xpt-forecasts-on-some-direct-approach-model-inputs",0,"","forecasting"],["“Dirty concepts” in AI alignment discourses, and some guesses for how to deal with them","Nora_Ammann and peckzy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bBicgqvwjPbaQrJJA/dirty-concepts-in-ai-alignment-discourses-and-some-guesses",0,"",""],["AI labs' requests for input","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DjAQMc9rAEyjkmwYb/ai-labs-requests-for-input",0,"",""],["Clarifying how misalignment can arise from scaling LLMs","Util","2023","blog","LessWrong","www.lesswrong.com/posts/wKZzLhhyADKqLAFan/clarifying-how-misalignment-can-arise-from-scaling-llms",0,"","scaling-laws"],["Supervised Program for Alignment Research (SPAR) at UC Berkeley: Spring 2023 summary","mic and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eW7YwLz548kDZaE6i/supervised-program-for-alignment-research-spar-at-uc",0,"",""],["Supervised Program for Alignment Research (SPAR) at UC Berkeley: Spring 2023 summary","mic and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/PXr38b64ECtFcn4Yq/supervised-program-for-alignment-research-spar-at-uc",0,"",""],["We can do better than DoWhatIMean","lukehmiles","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QQdmb3TjrvmB9Pfzp/we-can-do-better-than-dowhatimean",0,"",""],["Will AI kill everyone? Here's what the godfathers of AI have to say [RA video]","Writer and jai","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yxArBdibQejHEYT4F/will-ai-kill-everyone-here-s-what-the-godfathers-of-ai-have",0,"",""],["6 non-obvious mental health issues specific to AI safety","Igor Ivanov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Fj6wgJdDYuNP2FeD4/6-non-obvious-mental-health-issues-specific-to-ai-safety",0,"",""],["[Linkpost] Eric Schwitzgebel: AI systems must not confuse users about their sentience or moral status","Zachary Brown","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jWvgcLikfZj9MWhon/linkpost-eric-schwitzgebel-ai-systems-must-not-confuse-users",0,"",""],["An Overview of Catastrophic AI Risks: Summary","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9dNxz2kjNvPtiZjxj/an-overview-of-catastrophic-ai-risks-summary",0,"",""],["Managing risks of our own work","Beth Barnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fARMR2tiyCem8DD35/managing-risks-of-our-own-work",0,"","evals"],["AIのタイムライン ─ 提案されている論証と「専門家」の立ち位置","EA Japan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZcGLsL6kuHMGWsBjp/ainotaimurain-sareteiru-to-no-chi",0,"",""],["Autonomous replication and adaptation: an attempt at a concrete danger threshold","Hjalmar_Wijk","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vERGLBpDE8m5mpT6t/autonomous-replication-and-adaptation-an-attempt-at-a",0,"","evals"],["Corporate campaigns work: a key learning for AI Safety","Jamie_Harris","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zjmpFW3nBKwaBB5xr/corporate-campaigns-work-a-key-learning-for-ai-safety",0,"","governance"],["Launching Foresight Institute’s AI Grant for Underexplored Approaches to AI Safety – Apply for Funding!","elteerkers and Allison Duettmann","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EcKmt8ZJ3dcQBigna/launching-foresight-institute-s-ai-grant-for-underexplored",0,"",""],["Looking for judges for critiques of Alignment Plans","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/q7nWEbyW7tXwnKBe9/looking-for-judges-for-critiques-of-alignment-plans",0,"",""],["Making EA more inclusive, representative, and impactful in Africa","Ashura Batungwanayo and Hayley Martin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9cdntNDJQTS8dH5fh/making-ea-more-inclusive-representative-and-impactful-in",0,"",""],["The State of AI Governance in Africa: Musings from the Global South","Thaiya Jesse Wallace","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SsHBQLA6goqAmqvdS/the-state-of-ai-governance-in-africa-musings-from-the-global",0,"","governance policy"],["A Proof of Löb's Theorem using Computability Theory","jessicata","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/L6Ynch3CYMxXZkiq8/a-proof-of-loeb-s-theorem-using-computability-theory",0,"",""],["An argument for accelerating international AI governance research (part 1)","MattThinks","2023","blog","EA Forum","forum.effectivealtruism.org/posts/esAGKxupuLXhQ3bW5/an-argument-for-accelerating-international-ai-governance",0,"","governance policy"],["One example of how LLM propaganda attacks can hack the brain","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/fKNRHnxpjDLHnHdek/one-example-of-how-llm-propaganda-attacks-can-hack-the-brain",0,"","governance"],["Stampy's AI Safety Info - New Distillations #4 [July 2023]","markov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ttBKSopeo59AedWZA/stampy-s-ai-safety-info-new-distillations-4-july-2023",0,"",""],["Understanding and visualizing sycophancy datasets","Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/ZX9rgMfvZaxBseoYi/understanding-and-visualizing-sycophancy-datasets",0,"","sycophancy"],["A bill to prevent AI from hiring people instead of human enployers in NY","wes R","2023","blog","EA Forum","forum.effectivealtruism.org/posts/otsZNNLr2QygEM3Md/a-bill-to-prevent-ai-from-hiring-people-instead-of-human",0,"",""],["Am I taking crazy pills? Why aren't EAs advocating for a pause on AI capabilities?","anonymous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8HtkhGuyxgAXscCLz/am-i-taking-crazy-pills-why-aren-t-eas-advocating-for-a",0,"","governance"],["An Overview of Catastrophic AI Risks","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6WvnfKvF2i6mqp3za/an-overview-of-catastrophic-ai-risks",0,"","policy"],["Bio-x-AI policy: call for ideas from the Federation of American Scientists","Ben Stewart","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QnaDJvsxxzmwSN3Yh/bio-x-ai-policy-call-for-ideas-from-the-federation-of",0,"","policy"],["Credo AI is hiring for AI Gov Researcher & more!","IanEisenberg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LgKKwKgkWfdGES5xf/credo-ai-is-hiring-for-ai-gov-researcher-and-more",0,"","governance"],["Principles of Cyber-Physical Systems, Chapters 1-7,9","Rajeev Alur","2023","report","mitpress.mit.edu","mitpress.mit.edu/9780262548922/principles-of-cyber-physical-systems/",0,"",""],["Why some people disagree with the CAIS statement on AI","David_Moss and WillemSleegers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RYNtykh5xM467zRNj/why-some-people-disagree-with-the-cais-statement-on-ai",0,"",""],["$1,000 bounty for an AI Programme Lead recommendation","Cillian Crosson and Training for Good","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LdzWExZBLBBXScaog/usd1-000-bounty-for-an-ai-programme-lead-recommendation",0,"","policy forecasting"],["A short calculation about a Twitter poll","Ege Erdil","2023","blog","LessWrong","www.lesswrong.com/posts/ZdEhEeg9qnxwFgPMf/a-short-calculation-about-a-twitter-poll",0,"","theory"],["Decomposing independent generalizations in neural networks via Hessian analysis","Dmitry Vaintrob and Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/8ms977XZ2uJ4LnwSR/decomposing-independent-generalizations-in-neural-networks",0,"","interpretability"],["Import AI 336: Financialized AI; public and elite AI opinion; one million insects.","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-336-financialized-ai-public",0,"",""],["AGI is easier than robotaxis","Daniel Kokotajlo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/A5YQqDEz9QKGAZvn6/agi-is-easier-than-robotaxis",0,"",""],["AI-Relevant Regulation: CPSC","SWK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xtNa7bPehioFMjnqx/ai-relevant-regulation-cpsc",0,"","governance policy"],["Summary of “The Precipice” (2 of 4): We are a danger to ourselves","rileyharris","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ofC5eL88bC5Thjxoy/summary-of-the-precipice-2-of-4-we-are-a-danger-to-ourselves",0,"",""],["We Should Prepare for a Larger Representation of Academia in AI Safety","Leon Lang","2023","blog","LessWrong","www.lesswrong.com/posts/bjEDbjDp8xEAAE9yR/we-should-prepare-for-a-larger-representation-of-academia-in",0,"",""],["What do we know about Mustafa Suleyman's position on AI Safety?","Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JsjQRqvRc5pFmeSoj/what-do-we-know-about-mustafa-suleyman-s-position-on-ai",0,"",""],["Biological Anchors: The Trick that Might or Might Not Work","Scott Alexander","2023","blog","LessWrong","www.lesswrong.com/posts/NGkBfd8LTqcpbQn5Z/biological-anchors-the-trick-that-might-or-might-not-work",0,"","forecasting"],["AI Safety Concepts Writeup: WebGPT","Justis","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4ni3GBBzRRAgiksHT/ai-safety-concepts-writeup-webgpt",0,"",""],["When discussing AI risks, talk about capabilities, not intelligence","Vika","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JtuTQgp9Wnd6R6F5s/when-discussing-ai-risks-talk-about-capabilities-not",0,"",""],["A selection of some writings and considerations on the cause of artificial sentience","Raphaël_Pesah","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DP6GHJSNEbszBMF4s/a-selection-of-some-writings-and-considerations-on-the-cause",0,"",""],["Could We Automate AI Alignment Research?","Stephen McAleese","2023","blog","LessWrong","www.lesswrong.com/posts/zj7rjpAfuADkr7sqd/could-we-automate-ai-alignment-research-1",0,"","automated-alignment-research"],["Ilya Sutskever's thoughts on AI safety (July 2023): a transcript with my comments","mishka","2023","blog","LessWrong","www.lesswrong.com/posts/TpKktHS8GszgmMw4B/ilya-sutskever-s-thoughts-on-ai-safety-july-2023-a",0,"",""],["Seeking Input to AI Safety Book for non-technical audience","Darren McKee","2023","blog","LessWrong","www.lesswrong.com/posts/hi8MgnTjDCbh6kexs/seeking-input-to-ai-safety-book-for-non-technical-audience",0,"","governance"],["The positional embedding matrix and previous-token heads: how do they actually work?","AdamYedidia","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zRA8B2FJLtTYRgie6/the-positional-embedding-matrix-and-previous-token-heads-how",0,"","interpretability"],["UN Public Call for Nominations For High-level Advisory Body on Artificial Intelligence","vincentweisser","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zcthR42p6iHXDWYrw/un-public-call-for-nominations-for-high-level-advisory-body",0,"","policy"],["Update on cause area focus working group","Bastian_Stern","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3kMQTjtdWqkxGuWxB/update-on-cause-area-focus-working-group",0,"",""],["What Does a Marginal Grant at LTFF Look Like? Funding Priorities and Grantmaking Thresholds at the Long-Term Future Fund","Linch and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7RrjXQhGgAJiDLWYR/what-does-a-marginal-grant-at-ltff-look-like-funding",0,"","forecasting"],["4 types of AGI selection, and how to constrain them","Remmelt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Dkx7B2cSJMaLEzBKp/4-types-of-agi-selection-and-how-to-constrain-them",0,"",""],["Acausal Now: We could totally acausally bargain with aliens at our current tech level if desired","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/YgFbCWxzXYCpgzahe/acausal-now-we-could-totally-acausally-bargain-with-aliens",0,"","theory"],["Modulating sycophancy in an RLHF model via activation steering","NinaR","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/raoeNarFYCxxyKAop/modulating-sycophancy-in-an-rlhf-model-via-activation",0,"","rlhf sycophancy"],["Modulating sycophancy in an RLHF model via activation steering","Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/raoeNarFYCxxyKAop/modulating-sycophancy-in-an-rlhf-model-via-activation",0,"","rlhf sycophancy"],["When discussing AI risks, talk about capabilities, not intelligence","Victoria Krakovna","2023","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2023/08/09/when-discussing-ai-risks-talk-about-capabilities-not-intelligence/",0,"",""],["AISN #18: Challenges of Reinforcement Learning from Human Feedback, Microsoft’s Security Breach, and Conceptual Research on AI Safety","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/svD6fFGWvjsvCxjgM/aisn-18-challenges-of-reinforcement-learning-from-human",0,"","rlhf"],["Beginner's question about RLHF","FTPickle","2023","blog","LessWrong","www.lesswrong.com/posts/tksCZ7L8Xenk8GczJ/beginner-s-question-about-rlhf",0,"","rlhf"],["Ben Horowitz and others are spreading a \"regulation is bad\" view. Would it be useful to have a public bet on \"would Ben update his view if he had 1-1 with X-Risk researcher?\", and urge Ben to run such an experiment?","AntonOsika","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eWphfg7bcDrqfqgqF/ben-horowitz-and-others-are-spreading-a-regulation-is-bad",0,"","governance"],["Fundamentals of Global Priorities Research in Economics Syllabus","poliboni","2023","blog","EA Forum","forum.effectivealtruism.org/posts/g9gfXhNhLdJxSFBLW/fundamentals-of-global-priorities-research-in-economics",0,"",""],["Model Organisms of Misalignment: The Case for a New Pillar of Alignment Research","evhub and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ChDH335ckdvpxXaXX/model-organisms-of-misalignment-the-case-for-a-new-pillar-of-1",0,"","alignment-faking deception"],["OpenAI’s massive push to make superintelligence safe in 4 years or less (Jan Leike on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Kyh84cxzWcaKonHFG/openai-s-massive-push-to-make-superintelligence-safe-in-4",0,"",""],["Podcast (+transcript): Nathan Barnard on how US financial regulation can inform AI governance","Aaron Bergman and Nathan_Barnard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hASnoLMEFj3osCLKG/podcast-transcript-nathan-barnard-on-how-us-financial",0,"","evals governance policy"],["An interactive introduction to grokking and mechanistic interpretability","Adam Pearce and Asma Ghandeharioun","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/XpCnhaAQrssq8tJBG/an-interactive-introduction-to-grokking-and-mechanistic",0,"","interpretability mechanistic-interpretability"],["Optimisation Measures: Desiderata, Impossibility, Proposals","mattmacdermott and Alexander Gietelink Oldenziel","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JJnSHX33ShRmffNaR/optimisation-measures-desiderata-impossibility-proposals-2",0,"","agents theory"],["Studying Large Language Model Generalization with Influence Functions","Roger Grosse and 16 others","2023","paper","arXiv preprint","arxiv.org/abs/2308.03296",0,"",""],["Updates from Campaign for AI Safety","Jolyn Khoo and Nik Samoylov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gqumJhAKvy97Atww8/updates-from-campaign-for-ai-safety-3",0,"",""],["Rebooting AI Governance: An AI-Driven Approach to AI Governance","Max Reddel","2023","blog","LessWrong","www.lesswrong.com/posts/WLT3iajuTmwkCquSm/rebooting-ai-governance-an-ai-driven-approach-to-ai",0,"","governance forecasting"],["Safety-First Agents/Architectures Are a Promising Path to Safe AGI","Brendon_Wong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2hvYyzfWv4J3JLB8p/safety-first-agents-architectures-are-a-promising-path-to",0,"","interpretability agents"],["Yann LeCun on AGI and AI Safety","Chris_Leong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zfik4xESDyahRALKk/yann-lecun-on-agi-and-ai-safety",0,"",""],["An appeal to people who are smarter than me: please help me clarify my thinking about AI","bethhw","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BpxKj5P9dRBB4ged6/an-appeal-to-people-who-are-smarter-than-me-please-help-me",0,"",""],["Ground-Truth Label Imbalance Impairs Contrast-Consistent Search Performance","Tom Angsten and Ami Hays","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yB89JQdazhsDJhktH/ground-truth-label-imbalance-impairs-contrast-consistent-1",0,"","interpretability eliciting-latent-knowledge"],["Ground-Truth Label Imbalance Impairs the Performance of Contrast-Consistent Search (and Other Contrast-Pair-Based Unsupervised Methods)","Tom Angsten and Ami Hays","2023","blog","LessWrong","www.lesswrong.com/posts/yB89JQdazhsDJhktH/ground-truth-label-imbalance-impairs-the-performance-of-1",0,"","interpretability eliciting-latent-knowledge"],["Join AISafety.info's Writing & Editing Hackathon (Aug 25-28) (Prizes to be won!)","Siao Si and Stampy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZgbHXyushSdxxNjS2/join-aisafety-info-s-writing-and-editing-hackathon-aug-25-28",0,"",""],["[Linkpost] Multimodal Neurons in Pretrained Text-Only Transformers","Bogdan Ionut Cirstea","2023","blog","LessWrong","www.lesswrong.com/posts/wfhnDsavsoxnvGRKD/linkpost-multimodal-neurons-in-pretrained-text-only",0,"","interpretability"],["Apollo Research is hiring evals and interpretability engineers & scientists","mariushobbhahn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zcMhnfM9n6NKBaRnH/apollo-research-is-hiring-evals-and-interpretability",0,"","interpretability evals"],["Apollo Research is hiring evals and interpretability engineers & scientists","Marius Hobbhahn","2023","blog","LessWrong","www.lesswrong.com/posts/MrdFL38Zi3DwTDkKS/apollo-research-is-hiring-evals-and-interpretability",0,"","interpretability evals alignment-faking deception"],["AI #23: Fundamental Problems with RLHF","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/aKzwwKT2cy72awSyz/ai-23-fundamental-problems-with-rlhf",0,"","rlhf"],["Password-locked models: a stress case for capabilities evaluation","Fabien Roger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rZs6ddqNnW8LXuJqA/password-locked-models-a-stress-case-for-capabilities",0,"","evals"],["Training for Good is hiring (and why you should join us): AI Programme Lead and Operations Associate","Cillian Crosson and Training for Good","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iR3cwZgoQe3R47Lgr/training-for-good-is-hiring-and-why-you-should-join-us-ai",0,"","policy robustness"],["3 levels of threat obfuscation","HoldenKarnofsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HpzHjKjGQ4cKiY3jX/3-levels-of-threat-obfuscation",0,"","alignment-faking deception"],["3 levels of threat obfuscation","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hEwtb9Zjt5qwc2ygH/3-levels-of-threat-obfuscation",0,"",""],["Alignment Grantmaking is Funding-Limited Right Now [crosspost]","johnswentworth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BErC24s77fdo93ghi/alignment-grantmaking-is-funding-limited-right-now-crosspost",0,"",""],["Four part playbook for dealing with AI (Holden Karnofsky on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yumtQxSbwuDsfWqcb/four-part-playbook-for-dealing-with-ai-holden-karnofsky-on",0,"",""],["How many people are neartermist and have high P(doom)?","Sanjay","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DPfGxeWFLQaWEgBTj/how-many-people-are-neartermist-and-have-high-p-doom",0,"",""],["AI romantic partners will harm society if they go unregulated","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/FK8SwbcCq4HqvXmLv/ai-romantic-partners-will-harm-society-if-they-go",0,"","governance"],["AISN #17: Automatically Circumventing LLM Guardrails, the Frontier Model Forum, and Senate Hearing on AI Oversight","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hy48QDCDL7A7cGQ75/aisn-17-automatically-circumventing-llm-guardrails-the",0,"",""],["ARC Evals new report: Evaluating Language-Model Agents on Realistic Autonomous Tasks","Beth Barnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EPLk8QxETC5FEhoxK/arc-evals-new-report-evaluating-language-model-agents-on",0,"","evals agents"],["Artificially sentient beings: Moral, political, and legal issues","Fırat Akova","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ggom3PzSLS9wJrgBH/artificially-sentient-beings-moral-political-and-legal",0,"",""],["Confidence-Building Measures for Artificial Intelligence: Workshop proceedings","Sarah Shoker and Andrew Reddie","2023","blog","openai.com","openai.com/research/confidence-building-measures-for-artificial-intelligence",0,"",""],["Evaluating Superhuman Models with Consistency Checks","Daniel Paleka and Lukas Fluri","2023","blog","LessWrong","www.lesswrong.com/posts/WEjBvskFwBkczqjZZ/evaluating-superhuman-models-with-consistency-checks-1",0,"","evals"],["Riesgos Catastróficos Globales needs funding","Jaime Sevilla","2023","blog","EA Forum","forum.effectivealtruism.org/posts/h9unK57kLnmKdG6uq/riesgos-catastroficos-globales-needs-funding",0,"",""],["What is autonomy, and how does it lead to greater risk from AI?","Davidmanheim","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pbrJduve9kLA2yiZq/what-is-autonomy-and-how-does-it-lead-to-greater-risk-from",0,"","forecasting"],["EU’s AI ambitions at risk as US pushes to water down international treaty (linkpost)","mic","2023","blog","LessWrong","www.lesswrong.com/posts/uwr9bL8GA8uBmzbef/eu-s-ai-ambitions-at-risk-as-us-pushes-to-water-down",0,"","governance"],["How to find AI alignment researchers to collaborate with?","Florian Dietz","2023","blog","LessWrong","www.lesswrong.com/posts/h3usujzAcdMTetszs/how-to-find-ai-alignment-researchers-to-collaborate-with",0,"",""],["If AIs had subcortical brain simulation, would that solve the alignment problem?","Rainbow Affect","2023","blog","EA Forum","forum.effectivealtruism.org/posts/d7ocec7gNW3KNX6Nz/if-ais-had-subcortical-brain-simulation-would-that-solve-the",0,"",""],["Import AI 335: Synth data is a bad AI drug; Facebook changes the internet with LLaMa release; and Chinese researchers use AI to figure out chip design","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-335-synth-data-is-a-bad",0,"",""],["Is there any existing term summarizing non-scalable oversight methods in outer alignment?","Allen Shen","2023","blog","LessWrong","www.lesswrong.com/posts/gJnHebSag9Woz6raR/is-there-any-existing-term-summarizing-non-scalable",0,"","scalable-oversight"],["Open Problems and Fundamental Limitations of RLHF","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LqRD7sNcpkA9cmXLv/open-problems-and-fundamental-limitations-of-rlhf",0,"","rlhf"],["The “no sandbagging on checkable tasks” hypothesis","Joe Carlsmith","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/h7QETH7GMk9HcMnHH/the-no-sandbagging-on-checkable-tasks-hypothesis",0,"","sandbagging"],["The “no sandbagging on checkable tasks” hypothesis","Joe_Carlsmith","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MM22jJnQkLq2tPKHk/the-no-sandbagging-on-checkable-tasks-hypothesis",0,"","sandbagging"],["Thoughts on sharing information about language model capabilities","paulfchristiano","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fRSj2W4Fjje8rQWm9/thoughts-on-sharing-information-about-language-model",0,"","evals"],["Trading off compute in training and inference (Overview)","Pablo Villalobos","2023","blog","LessWrong","www.lesswrong.com/posts/hDuoHuBjxGue7fwhJ/trading-off-compute-in-training-and-inference-overview",0,"","governance"],["Watermarking considered overrated?","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pvE6NdPoMCyk55Lxn/watermarking-considered-overrated",0,"",""],["README.docx","Vael Gates","2023","report","drive.google.com","drive.google.com/file/d/1n6_WYIQytoyNAIXZrE0b0cQBlQ2CiZeg/view",0,"",""],["Shutting down AI Safety Support","JJ Hepburn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Bjr6FXvnKqb37uMPP/shutting-down-ai-safety-support",0,"",""],["Fundamentals of Fatal Risks","Aino","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yBavn9rtThLFTfLoz/fundamentals-of-fatal-risks",0,"","policy theory"],["Announcing the ITAM AI Futures Fellowship","AmAristizabal and Jaime Andres Fernandez","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FvgQjicdSk6S7xQvC/announcing-the-itam-ai-futures-fellowship",0,"","policy"],["Introductory Textbook to Vision Models Interpretability","jeanne_ and Charbel-Raphaël","2023","blog","LessWrong","www.lesswrong.com/posts/XZfJvxZqfbLfN6pKh/introductory-textbook-to-vision-models-interpretability",0,"","interpretability"],["Mech Interp Puzzle 2: Word2Vec Style Embeddings","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hZGoeGdJsnzJbQJMp/mech-interp-puzzle-2-word2vec-style-embeddings",0,"","interpretability"],["Reducing sycophancy and improving honesty via activation steering","NinaR","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zt6hRsDE84HeBKh7E/reducing-sycophancy-and-improving-honesty-via-activation",0,"","sycophancy"],["Reducing sycophancy and improving honesty via activation steering","Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/zt6hRsDE84HeBKh7E/reducing-sycophancy-and-improving-honesty-via-activation",0,"","sycophancy"],["US Congress introduces CREATE AI Act for establishing National AI Research Resource","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qEKqhZFsx5zwyf9Mu/us-congress-introduces-create-ai-act-for-establishing",0,"","evals governance policy"],["Visible loss landscape basins don't correspond to distinct algorithms","Mikhail Samin","2023","blog","LessWrong","www.lesswrong.com/posts/muLN8GRBdB8NLLX36/visible-loss-landscape-basins-don-t-correspond-to-distinct",0,"","interpretability"],["Visit Mexico City in January & February to interact with the AI Futures Fellowship","AmAristizabal and Jaime Andres Fernandez","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MN34Pd6gCeHPgnMwH/visit-mexico-city-in-january-and-february-to-interact-with",0,"",""],["When can we trust model evaluations?","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dBmfb76zx6wjPsBC7/when-can-we-trust-model-evaluations",0,"","evals"],["Animal Advocacy in the Age of AI","Constance Li and Nicholas Kees Dupuis","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oGoP4LjSZAsYfcF3N/animal-advocacy-in-the-age-of-ai-1",0,"",""],["AXRP Episode 23 - Mechanistic Anomaly Detection with Mark Xu","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jghQXYvifXxhzkhHM/axrp-episode-23-mechanistic-anomaly-detection-with-mark-xu",0,"","interpretability monitoring"],["AXRP Episode 24 - Superalignment with Jan Leike","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bsNXqHgiDA6dAKNun/axrp-episode-24-superalignment-with-jan-leike",0,"",""],["Discussing AI-Human Collaboration Through Fiction: The Story of Laika and GPT-∞","Laika","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gomS2ocBzXJA2Mg3w/discussing-ai-human-collaboration-through-fiction-the-story",0,"",""],["Partial Transcript of Recent Senate Hearing Discussing AI X-Risk","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/67zFQT4GeJdgvdFuk/partial-transcript-of-recent-senate-hearing-discussing-ai-x",0,"","governance policy"],["Preference Aggregation as Bayesian Inference","beren","2023","blog","LessWrong","www.lesswrong.com/posts/KCHQj2ZDAuSZb4Nif/preference-aggregation-as-bayesian-inference",0,"","theory"],["AGI Takeoff dynamics - Intelligence vs Quantity explosion","EdoArad","2023","blog","EA Forum","forum.effectivealtruism.org/posts/S4f2wvw6HBioWzXCy/agi-takeoff-dynamics-intelligence-vs-quantity-explosion",0,"","forecasting"],["Apply to CEEALAR to do AGI moratorium work","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AJJTRmW7zhvXrmD5s/apply-to-ceealar-to-do-agi-moratorium-work",0,"","governance"],["EleutherAI's Thoughts on the EU AI Act","Aviya Skowron and Stella Biderman","2023","blog","blog.eleuther.ai","blog.eleuther.ai/eu-aia/",0,"",""],["Evaluating the Moral Beliefs Encoded in LLMs Warning: This paper contains moral scenarios which are controversial and offensive in nature.","Nino Scherrer","2023","paper","arXiv preprint","arxiv.org/abs/2307.14324",0,"","evals"],["Existential risk from AI and what DC could do about it (Ezra Klein on the 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6ugipGgCiYsBzbMwt/existential-risk-from-ai-and-what-dc-could-do-about-it-ezra-1",0,"",""],["Frontier Model Security","Vaniver","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/oNGCgNag2zSMqg2Z7/frontier-model-security",0,"",""],["Meta-level adversarial evaluation of oversight techniques might allow robust measurement of their adequacy","Buck and ryan_greenblatt","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MbWWKbyD5gLhJgfwn/meta-level-adversarial-evaluation-of-oversight-techniques-1",0,"","evals"],["\"The Universe of Minds\" - call for reviewers (Seeds of Science)","rogersbacon1","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BbpYq9iwGzC4YBfMk/the-universe-of-minds-call-for-reviewers-seeds-of-science",0,"",""],["[Linkpost] My attempt at trying to summarize 'Intro to ML Safety'","Arjun Yadav","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kPPneWBzDhuRoXLq5/linkpost-my-attempt-at-trying-to-summarize-intro-to-ml",0,"",""],["AI Safety Hub Serbia Soft Launch","Dušan D. Nešić (Dushan)","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7gL7CFBmybjAjJvAw/ai-safety-hub-serbia-soft-launch",0,"",""],["AI Safety Hub Serbia Soft Launch","DusanDNesic","2023","blog","LessWrong","www.lesswrong.com/posts/CmvkoyTq49tFkSGFF/ai-safety-hub-serbia-soft-launch",0,"",""],["AISN #16: White House Secures Voluntary Commitments from Leading AI Labs and Lessons from Oppenheimer","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7yK5fB7y3bb8dEMED/aisn-16-white-house-secures-voluntary-commitments-from",0,"",""],["Carl Shulman on AI takeover mechanisms (& more): Part II of Dwarkesh Patel interview for The Lunar Society","alejandro","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kHqfZczkcp5Wp4yvW/carl-shulman-on-ai-takeover-mechanisms-and-more-part-ii-of",0,"","governance forecasting"],["How LLMs are and are not myopic","janus","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/c68SJsBpiAxkPwRHj/how-llms-are-and-are-not-myopic",0,"",""],["Should you work at a leading AI lab? (including in non-safety roles)","Benjamin Hilton and 80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Hyupm7mPgoXNu9PEW/should-you-work-at-a-leading-ai-lab-including-in-non-safety",0,"",""],["Summary of posts on XPT forecasts on AI risk and timelines","Forecasting Research Institute and rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HXoHFHCSFX7Cxn7hC/summary-of-posts-on-xpt-forecasts-on-ai-risk-and-timelines",0,"","forecasting"],["Task decomposition for scalable oversight (AGISF Distillation)","Charbel-Raphaël","2023","blog","LessWrong","www.lesswrong.com/posts/FFz6H35Gy6BArHxkc/task-decomposition-for-scalable-oversight-agisf-distillation",0,"","scalable-oversight"],["Towards evidence gap-maps for AI safety","dEAsign","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZD7KxnqfXR7cyouxP/towards-evidence-gap-maps-for-ai-safety",0,"","governance"],["[Crosspost] An AI Pause Is Humanity's Best Bet For Preventing Extinction (TIME)","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/D4khSueGA4Trebkks/crosspost-an-ai-pause-is-humanity-s-best-bet-for-preventing",0,"","governance policy"],["[Crosspost] An AI Pause Is Humanity's Best Bet For Preventing Extinction (TIME)","otto.barten","2023","blog","LessWrong","www.lesswrong.com/posts/bR8zWoYS9zfha8Hzo/crosspost-an-ai-pause-is-humanity-s-best-bet-for-preventing",0,"","governance"],["[link post] AI Should Be Terrified of Humans","BrianK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EhPKbX5JkwxvhfGhC/link-post-ai-should-be-terrified-of-humans",0,"",""],["Asterisk Magazine Issue 03: AI","alejandro","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qNKHumeLwTamkD5ED/asterisk-magazine-issue-03-ai",0,"","evals governance compute-governance forecasting"],["Open problems in activation engineering","TurnTrout and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JMebqicMD6azB8MwK/open-problems-in-activation-engineering",0,"","interpretability"],["Slowing down AI progress is an underexplored alignment strategy","Norman Borlaug","2023","blog","LessWrong","www.lesswrong.com/posts/b7JXJWY7R2jNtHerP/slowing-down-ai-progress-is-an-underexplored-alignment",0,"","governance"],["XPT forecasts on (some) biological anchors inputs","Forecasting Research Institute and rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ccw9v9giKxg8nyLhp/xpt-forecasts-on-some-biological-anchors-inputs",0,"","forecasting"],["My favorite AI governance research this year so far","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/gzJ7QNhd3tCLkbmYC/my-favorite-ai-governance-research-this-year-so-far",0,"","governance"],["QAPR 5: grokking is maybe not *that* big a deal?","Quintin Pope","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GpSzShaaf8po4rcmA/qapr-5-grokking-is-maybe-not-that-big-a-deal",0,"",""],["Supplementary Alignment Insights Through a Highly Controlled Shutdown Incentive","Justausername","2023","blog","LessWrong","www.lesswrong.com/posts/Yc6KdHYFMXwzPdZAX/supplementary-alignment-insights-through-a-highly-controlled",0,"","alignment-faking deception"],["AI-Relevant Regulation: Insurance in Safety-Critical Industries","SWK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KDAXeG9Cz64gBDYXC/ai-relevant-regulation-insurance-in-safety-critical",0,"","governance policy"],["Compute Thresholds: proposed rules to mitigate risk of a “lab leak” accident during AI training runs","davidad","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zfk6faYvcf5Ht7xDx/compute-thresholds-proposed-rules-to-mitigate-risk-of-a-lab",0,"","governance compute-governance"],["Could someone help me understand why it's so difficult to solve the alignment problem?","Jadon Schmitt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CmvqwYfuzR6E5HPfd/could-someone-help-me-understand-why-it-s-so-difficult-to",0,"",""],["Examples of Prompts that Make GPT-4 Output Falsehoods","scasper and Luke Bailey","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iNvWCTpyEd4zTqjjv/examples-of-prompts-that-make-gpt-4-output-falsehoods-1",0,"",""],["Australians call for AI safety to be taken seriously","AlexanderSaeri","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9btFvtGwkufZpC7Yu/australians-call-for-ai-safety-to-be-taken-seriously",0,"","governance policy"],["BCIs and the ecosystem of modular minds","beren","2023","blog","LessWrong","www.lesswrong.com/posts/d74pb97TAqNKwJkc5/bcis-and-the-ecosystem-of-modular-minds",0,"","forecasting"],["Excerpts from \"Majority Leader Schumer Delivers Remarks To Launch SAFE Innovation Framework For Artificial Intelligence At CSIS\"","Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HeJ5BB9wh2TZ5ZrYn/excerpts-from-majority-leader-schumer-delivers-remarks-to",0,"",""],["GPT-2's positional embedding matrix is a helix","AdamYedidia","2023","blog","LessWrong","www.lesswrong.com/posts/qvWP3aBDBaqXvPNhS/gpt-2-s-positional-embedding-matrix-is-a-helix",0,"","interpretability"],["Linkpost: 7 A.I. Companies Agree to Safeguards After Pressure From the White House","MHR","2023","blog","EA Forum","forum.effectivealtruism.org/posts/74CkwGxmXaevwzhNG/linkpost-7-a-i-companies-agree-to-safeguards-after-pressure",0,"","governance policy"],["News : Biden-⁠Harris Administration Secures Voluntary Commitments from Leading Artificial Intelligence Companies to Manage the Risks Posed by AI","Jonathan Claybrough","2023","blog","LessWrong","www.lesswrong.com/posts/K49G5XSinhoAknncQ/news-biden-harris-administration-secures-voluntary",0,"","governance"],["Priorities for the UK Foundation Models Taskforce","Andrea_Miotti","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YAgQysWqoBJ7eA7Np/priorities-for-the-uk-foundation-models-taskforce",0,"","governance"],["Reward Hacking from a Causal Perspective","tom4everitt and 5 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aw5nqamqtnDnW8w9u/reward-hacking-from-a-causal-perspective",0,"","reward-hacking"],["Training Process Transparency through Gradient Interpretability: Early experiments on toy language models","robertzk and evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DtkA5jysFZGv7W4qP/training-process-transparency-through-gradient",0,"","interpretability"],["What do XPT forecasts tell us about AI timelines?","rosehadshar and Forecasting Research Institute","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KGGDduXSwZQTQJ9xc/what-do-xpt-forecasts-tell-us-about-ai-timelines",0,"","forecasting"],["All AGI Safety questions welcome (especially basic ones) [July 2023]","smallsilo","2023","blog","LessWrong","www.lesswrong.com/posts/xroYDAE6EoisrSFZf/all-agi-safety-questions-welcome-especially-basic-ones-july-2",0,"",""],["Does Circuit Analysis Interpretability Scale? Evidence from Multiple Choice Capabilities in Chinchilla","Neel Nanda and 6 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Av3frxNy3y3i2kpaa/does-circuit-analysis-interpretability-scale-evidence-from",0,"","interpretability mechanistic-interpretability"],["Epoch is hiring an ML Hardware Researcher","merilalama","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iekFPDBHusqqvSmsy/epoch-is-hiring-an-ml-hardware-researcher",0,"","forecasting"],["Even Superhuman Go AIs Have Surprising Failure Modes","AdamGleave and 7 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DCL3MmMiPsuMxP45a/even-superhuman-go-ais-have-surprising-failure-modes",0,"","agents robustness"],["Should we nationalize AI development?","Jadon Schmitt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SezmJHRmdxufBzEmC/should-we-nationalize-ai-development",0,"","governance policy"],["Speculative inferences about path dependence in LLM supervised fine-tuning from results on linear mode connectivity and model souping","RobertKirk","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rYdRiaA3cuioJxmBv/speculative-inferences-about-path-dependence-in-llm",0,"","scaling-laws"],["The Dilemma of Ultimate Technology","Aino","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fwGevCo3bvypymhwb/the-dilemma-of-ultimate-technology",0,"",""],["AISN#15: China and the US take action to regulate AI, results from a tournament forecasting AI risk, updates on xAI’s plan, and Meta releases its open-source and commercially available Llama 2","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LYWXDKfjyJquYF9Gm/aisn-15-china-and-the-us-take-action-to-regulate-ai-results",0,"","forecasting"],["Alignment Grantmaking is Funding-Limited Right Now","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/SbC7duHNDHkd3PkgG/alignment-grantmaking-is-funding-limited-right-now",0,"",""],["All AGI Safety questions welcome (especially basic ones) [July 2023]","Siao Si and Stampy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vGfJnwq6X7hwhG3wy/all-agi-safety-questions-welcome-especially-basic-ones-july",0,"",""],["An Introduction to Critiques of prominent AI safety organizations","Omega","2023","blog","EA Forum","forum.effectivealtruism.org/posts/N4LKrktopDs5Qdqgn/an-introduction-to-critiques-of-prominent-ai-safety",0,"",""],["Desiderata for an AI","Nathan Helm-Burger","2023","blog","LessWrong","www.lesswrong.com/posts/ZxHfuCyfAiHAy9Mds/desiderata-for-an-ai",0,"","interpretability agents robustness"],["Hedonic Loops and Taming RL","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3mwfyLpnYqhqvprbb/hedonic-loops-and-taming-rl",0,"","interpretability instrumental-convergence"],["Incident reporting for AI safety","Zach Stein-Perlman and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qkK5ejystp8GCJ3vC/incident-reporting-for-ai-safety",0,"",""],["Thoughts on yesterday’s UN Security Council meeting on AI","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DNm5sbFogr9wvDasH/thoughts-on-yesterday-s-un-security-council-meeting-on-ai",0,"","governance policy"],["Updates from Campaign for AI Safety","Jolyn Khoo and Nik Samoylov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FicNtafsLGnFkf7yA/updates-from-campaign-for-ai-safety-1",0,"",""],["Using predictors in corrigible systems","porby","2023","blog","LessWrong","www.lesswrong.com/posts/LR8yhJCBffky8X3Az/using-predictors-in-corrigible-systems",0,"",""],["What do XPT forecasts tell us about AI risk?","Forecasting Research Institute and rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/K2xQrrXn5ZSgtntuT/what-do-xpt-forecasts-tell-us-about-ai-risk-1",0,"","forecasting"],["AI Impacts Quarterly Newsletter, Apr-Jun 2023","Harlan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Bmjucdecv3p5smNC8/ai-impacts-quarterly-newsletter-apr-jun-2023",0,"",""],["Five Years of Rethink Priorities: Impact, Future Plans, Funding Needs (July 2023)","Rethink Priorities","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7QyemcXLaxNicLNNa/five-years-of-rethink-priorities-impact-future-plans-funding",0,"",""],["I'm interviewing Jan Leike, co-lead of OpenAI's new Superalignment project. What should I ask him?","Robert_Wiblin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rdxHxZYusMsihf8Qk/i-m-interviewing-jan-leike-co-lead-of-openai-s-new",0,"",""],["Measuring and Improving the Faithfulness of Model-Generated Reasoning","Ansh Radhakrishnan and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BKvJNzALpxS3LafEs/measuring-and-improving-the-faithfulness-of-model-generated",0,"",""],["Meta announces Llama 2; \"open sources\" it for commercial use","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9rdpdDerangjYeQGW/meta-announces-llama-2-open-sources-it-for-commercial-use",0,"",""],["Simple alignment plan that maybe works","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/mpj398Dy2NMB66hLY/simple-alignment-plan-that-maybe-works",0,"",""],["Still no Lie Detector for LLMs","Whispermute and ben_levinstein","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bCQbSFrnnAk7CJNpM/still-no-lie-detector-for-llms",0,"","interpretability eliciting-latent-knowledge"],["Tiny Mech Interp Projects: Emergent Positional Embeddings of Words","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ln7D2aYgmPgjhpEeA/tiny-mech-interp-projects-emergent-positional-embeddings-of",0,"","interpretability"],["Train for incorrigibility, then reverse it (Shutdown Problem Contest Submission)","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MeigEG9KgJp9jFFuR/train-for-incorrigibility-then-reverse-it-shutdown-problem",0,"",""],["A fictional AI law laced w/ alignment theory","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BZiJ7C6cxrHsDbHsh/a-fictional-ai-law-laced-w-alignment-theory",0,"","governance"],["A fictional AI law laced w/ alignment theory","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/7wCeeqXYksnBeFSbx/a-fictional-ai-law-laced-w-alignment-theory",0,"","governance"],["AutoInterpretation Finds Sparse Coding Beats Alternatives","Hoagy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ursraZGcpfMjCXtnn/autointerpretation-finds-sparse-coding-beats-alternatives",0,"","interpretability"],["Developing reliable AI tools for healthcare","Krishnamurthy (Dj) Dvijotham and Taylan Cemgil on behalf of the CoDoC team","2023","blog","deepmind.com","www.deepmind.com/blog/codoc-developing-reliable-ai-tools-for-healthcare",0,"",""],["Eliciting responses to Marc Andreessen's \"Why AI Will Save the World\"","Coleman@21stTalks","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CbBcrqkPCEc2tSgyq/eliciting-responses-to-marc-andreessen-s-why-ai-will-save",0,"",""],["New career review: AI safety technical research","Benjamin Hilton and 80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SAvkXAwrzdhecAaCj/new-career-review-ai-safety-technical-research",0,"",""],["The shape of AGI: Cartoons and back of envelope","boazbarak","2023","blog","LessWrong","www.lesswrong.com/posts/bku9odAYPyQwHqzCo/the-shape-of-agi-cartoons-and-back-of-envelope",0,"","forecasting"],["Thoughts on “Process-Based Supervision”","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/D4gEDdqWrgDPMtasc/thoughts-on-process-based-supervision-1",0,"",""],["What we can learn from stress testing for AI regulation","Nathan_Barnard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EehuKzAvEbDbXxTwJ/what-we-can-learn-from-stress-testing-for-ai-regulation",0,"","governance"],["A simple way of exploiting AI's coming economic impact may be highly-impactful","kuira","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tmxNQJ48SWEWXSt4i/a-simple-way-of-exploiting-ai-s-coming-economic-impact-may",0,"","forecasting"],["Activation adding experiments with llama-7b","Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/w9yKQzyhsLJEZhvg9/activation-adding-experiments-with-llama-7b",0,"","interpretability"],["An upcoming US Supreme Court case may impede AI governance efforts","NickGabs","2023","blog","LessWrong","www.lesswrong.com/posts/oSrfAYpGLXAeZYmvY/an-upcoming-us-supreme-court-case-may-impede-ai-governance",0,"","governance"],["Embracing the automated future","Arjun Khemani","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MxnCf9qBTFygKnziF/embracing-the-automated-future",0,"",""],["Even briefer summary of ai-plans.com","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/76D2sKhnKNMyJ8YbX/even-briefer-summary-of-ai-plans-com",0,"",""],["Less activations can result in high corrigibility?","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/Krc8HqJYLFNZYvbEr/less-activations-can-result-in-high-corrigibility",0,"","interpretability"],["Mech Interp Puzzle 1: Suspiciously Similar Embeddings in GPT-Neo","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/eLNo7b56kQQerCzp2/mech-interp-puzzle-1-suspiciously-similar-embeddings-in-gpt",0,"","interpretability"],["Runaway Optimizers in Mind Space","silentbob","2023","blog","LessWrong","www.lesswrong.com/posts/Mb96sKbqERmcx22hw/runaway-optimizers-in-mind-space",0,"",""],["Scaling and Sustaining Standards: A Case Study on the Basel Accords","Conrad K.","2023","blog","LessWrong","www.lesswrong.com/posts/cLGRGKDrhhNP6Jgub/scaling-and-sustaining-standards-a-case-study-on-the-basel",0,"","governance"],["AI-Relevant Regulation: CERN","SWK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PJxkdzTTYDyrRT99M/ai-relevant-regulation-cern",0,"","governance policy"],["AI-Relevant Regulation: IAEA","SWK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pX63E56uNkQgHJvx6/ai-relevant-regulation-iaea",0,"","governance policy"],["Cambridge AI Safety Hub is looking for full- or part-time organisers","hannah","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ijeBndPQdx8kCcM2R/cambridge-ai-safety-hub-is-looking-for-full-or-part-time",0,"",""],["Introducción al Riesgo Existencial de Inteligencia Artificial","david.friva","2023","blog","LessWrong","www.lesswrong.com/posts/8kbQaxveLyvyvxwcr/introduccion-al-riesgo-existencial-de-inteligencia",0,"",""],["Only a hack can solve the shutdown problem","dp","2023","blog","LessWrong","www.lesswrong.com/posts/Yh9nkqfoSs2KGfetf/only-a-hack-can-solve-the-shutdown-problem",0,"",""],["Robustness of Model-Graded Evaluations and Automated Interpretability","Simon Lermen and viluon","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZbjyCuqpwCMMND4fv/robustness-of-model-graded-evaluations-and-automated",0,"","interpretability evals automated-alignment-research robustness"],["Simplified bio-anchors for upper bounds on AI timelines","Fabien Roger","2023","blog","LessWrong","www.lesswrong.com/posts/FsfP3e7ZspCPuwaRA/simplified-bio-anchors-for-upper-bounds-on-ai-timelines",0,"","forecasting"],["Why was the AI Alignment community so unprepared for this moment?","Ras1513","2023","blog","LessWrong","www.lesswrong.com/posts/6jytXo5HmR9HLvkzu/why-was-the-ai-alignment-community-so-unprepared-for-this",0,"","governance"],["AI Risk and Survivorship Bias - How Andreessen and LeCun got it wrong","stepanlos","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yQHzdmXa7KBB52fBz/ai-risk-and-survivorship-bias-how-andreessen-and-lecun-got",0,"","governance"],["Gearing Up for Long Timelines in a Hard World","Dalcy Bremin","2023","blog","LessWrong","www.lesswrong.com/posts/f4NrqEKMsnRKdjtpx/gearing-up-for-long-timelines-in-a-hard-world",0,"","agents forecasting theory"],["New DeepMind report on institutions for global AI governance","finm","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rXLazPKnm7PrwGAs6/new-deepmind-report-on-institutions-for-global-ai-governance",0,"","governance"],["Book Review: Oryx and Crake","Benny Smith","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HPZwbWzMGCCr5iu2c/book-review-oryx-and-crake",0,"",""],["Instrumental Convergence to Complexity Preservation","Macro Flaneur","2023","blog","LessWrong","www.lesswrong.com/posts/WEPtBsJJKyfqjkBKc/instrumental-convergence-to-complexity-preservation",0,"","instrumental-convergence"],["Tetlock on low AI xrisk","TeddyW","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JsnhfXsNgi3GxfScL/tetlock-on-low-ai-xrisk",0,"","forecasting"],["What criterion would you use to select companies likely to cause AI doom?","amaury lorin","2023","blog","LessWrong","www.lesswrong.com/posts/TKKLL9Y4iCA6RMg8b/what-criterion-would-you-use-to-select-companies-likely-to",0,"","governance"],["What new psychology research could best promote AI safety & alignment research?","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sm65sDR6r4BmK7PGM/what-new-psychology-research-could-best-promote-ai-safety",0,"",""],["Winners of AI Alignment Awards Research Contest","Akash and Olivia Jimenez","2023","blog","LessWrong","www.lesswrong.com/posts/zFoAAD7dfWdczxoLH/winners-of-ai-alignment-awards-research-contest",0,"",""],["[Linkpost] NY Times Feature on Anthropic","Garrison","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jPW3jgfYPBrwHEbog/linkpost-ny-times-feature-on-anthropic",0,"",""],["A transcript of the TED talk by Eliezer Yudkowsky","Mikhail Samin","2023","blog","LessWrong","www.lesswrong.com/posts/kXHBz2Z5BBgaBiazf/a-transcript-of-the-ted-talk-by-eliezer-yudkowsky",0,"",""],["AISN#14: OpenAI’s ‘Superalignment’ team, Musk’s xAI launches, and developments in military AI use","Center for AI Safety and Dan H","2023","blog","EA Forum","forum.effectivealtruism.org/posts/k7QW3F4GzSd6QYKp9/aisn-14-openai-s-superalignment-team-musk-s-xai-launches-and",0,"",""],["Alignment Megaprojects: You're Not Even Trying to Have Ideas","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/5xrkjHCvCeeDtHa5g/alignment-megaprojects-you-re-not-even-trying-to-have-ideas",0,"",""],["An Overview of the AI Safety Funding Situation","Stephen McAleese","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XdhwXppfqrpPL2YDX/an-overview-of-the-ai-safety-funding-situation",0,"","forecasting"],["Announcing the AI Fables Writing Contest!","Daystar Eld","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gYxY5Mr2srBnrbuaT/announcing-the-ai-fables-writing-contest",0,"",""],["Betting on Logic","Sylvester Kollin","2023","blog","LessWrong","www.lesswrong.com/posts/XAma8pvsKGJZsNLDt/betting-on-logic",0,"","theory"],["Could unions be an underrated driver for AI safety policy?","Dunning K.","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rdWKqSzia2yBz8Z7i/could-unions-be-an-underrated-driver-for-ai-safety-policy",0,"","policy"],["Eric Michaud on the Quantization Model of Neural Scaling, Interpretability and Grokking","Michaël Trazzi","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/xKT5oTXDCJNramc7y/eric-michaud-on-the-quantization-model-of-neural-scaling",0,"","interpretability"],["Goal-Direction for Simulated Agents","Raymond D","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hmPCPyr6JFLEigHJx/goal-direction-for-simulated-agents",0,"","agents"],["How I Learned To Stop Worrying And Love The Shoggoth","Peter Merel","2023","blog","LessWrong","www.lesswrong.com/posts/zYJMf7QoaNahccxrp/how-i-learned-to-stop-worrying-and-love-the-shoggoth",0,"","automated-alignment-research governance forecasting"],["Report on modeling evidential cooperation in large worlds","Johannes Treutlein","2023","blog","LessWrong","www.lesswrong.com/posts/mKnbHwRc7mXtENNEm/report-on-modeling-evidential-cooperation-in-large-worlds",0,"","theory"],["The Opt-In Revolution — My vision of a positive future with ASI (An experiment with LLM storytelling)","Tachikoma","2023","blog","LessWrong","www.lesswrong.com/posts/sGBszCBKp6roEd8v5/the-opt-in-revolution-my-vision-of-a-positive-future-with",0,"","forecasting"],["Towards Developmental Interpretability","Jesse Hoogland and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TjaeCWvLZtEDAS5Ex/towards-developmental-interpretability",0,"","interpretability"],["What does the launch of x.ai mean for AI Safety?","Chris_Leong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fccbYpCvTrBFMGmfu/what-does-the-launch-of-x-ai-mean-for-ai-safety",0,"",""],["(How) Is technical AI Safety research being evaluated?","JohnSnow","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WJMa3XeZMuukAm9Lb/how-is-technical-ai-safety-research-being-evaluated",0,"","evals"],["AI Wellbeing","Simon and cdkg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4vphSKe9aSSGuQRap/ai-wellbeing",0,"",""],["Disincentivizing deception in mesa optimizers with Model Tampering","martinkunev","2023","blog","LessWrong","www.lesswrong.com/posts/jEXfacKpuy87vBYWe/disincentivizing-deception-in-mesa-optimizers-with-model",0,"","alignment-faking deception"],["How to regulate cutting-edge AI models (Markus Anderljung on The 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HCdxb2hqnKE3pWs73/how-to-regulate-cutting-edge-ai-models-markus-anderljung-on",0,"",""],["OpenAI Launches Superalignment Taskforce","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/NSZhadmoYdjRKNq6X/openai-launches-superalignment-taskforce",0,"","automated-alignment-research"],["What is the most convincing article, video, etc. making the case that AI is an X-Risk","Jordan Arel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vhGQHwc5pKpBiaqAn/what-is-the-most-convincing-article-video-etc-making-the",0,"",""],["Arguments against existential risk from AI, part 2","Nina Rimsky","2023","blog","LessWrong","www.lesswrong.com/posts/q5GYAyXETnBATNCSw/arguments-against-existential-risk-from-ai-part-2",0,"",""],["Consciousness as a conflationary alliance term","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KpD2fJa6zo8o2MBxg/consciousness-as-a-conflationary-alliance-term",0,"",""],["Consider Joining the UK Foundation Model Taskforce","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/xgXcZQd5eqMqpAw3i/consider-joining-the-uk-foundation-model-taskforce",0,"","governance"],["Cost-effectiveness of professional field-building programs for AI safety research","Center for AI Safety","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7kFPFYQSY7ZttoveS/cost-effectiveness-of-professional-field-building-programs",0,"","forecasting"],["Cost-effectiveness of student programs for AI safety research","Center for AI Safety","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zYSAFtjasxsfm3nmh/cost-effectiveness-of-student-programs-for-ai-safety",0,"","forecasting"],["Do you think the probability of future AI sentience(suffering) is >0.1%? Why?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HsRWX2T6fBHXRyMaP/do-you-think-the-probability-of-future-ai-sentience",0,"",""],["GPT-7: The Tale of the Big Computer (An Experimental Story)","Justin Bullock","2023","blog","LessWrong","www.lesswrong.com/posts/MWnB22utwmPzt8zAG/gpt-7-the-tale-of-the-big-computer-an-experimental-story",0,"","governance"],["Import AI 334: Better distillation; the UK's AI taskforce; money and AI","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-334-better-distillation",0,"",""],["Incentives from a causal perspective","tom4everitt and 5 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xm4vSHaKAmfRvgBgi/incentives-from-a-causal-perspective",0,"","power-seeking"],["Infographics report risk management of Artificial Intelligence in Spain","JorgeTorresC and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hXjBJLqxdG5FNn3ps/infographics-report-risk-management-of-artificial",0,"",""],["Is the Endowment Effect Due to Incomparability?","Kevin Dorst","2023","blog","LessWrong","www.lesswrong.com/posts/MrcTzbYeZ3xnh9mGj/is-the-endowment-effect-due-to-incomparability",0,"",""],["Modeling the impact of AI safety field-building programs","Center for AI Safety","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Ykqh8ku7NHN9CGkdC/modeling-the-impact-of-ai-safety-field-building-programs",0,"","forecasting"],["Open-minded updatelessness","Nicolas Macé and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uPWDwFJnxLaDiyv4M/open-minded-updatelessness",0,"","theory"],["Urgent Need for Refinancing","Tobias W. Kaiser","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XHmWPCgXu7aGstTnZ/urgent-need-for-refinancing",0,"",""],["“Reframing Superintelligence” + LLMs + 4 years","Eric Drexler","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LxNwBNxXktvzAko65/reframing-superintelligence-llms-4-years",0,"","forecasting"],["epistemic range","Tamsin Leake","2023","blog","carado.moe","carado.moe/epistemic-range.html",0,"",""],["Some basics of the hypercompetence theory of government","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/pfL6sAjMfRsZjyjsZ/some-basics-of-the-hypercompetence-theory-of-government",0,"","governance"],["\"Concepts of Agency in Biology\" (Okasha, 2023) - Brief Paper Summary","Nora_Ammann","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/c27yRmcBxC6txibWW/concepts-of-agency-in-biology-okasha-2023-brief-paper",0,"",""],["Announcing AI Alignment workshop at the ALIFE 2023 conference","Rory Greig","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SQ6quv4aRvpDcg2vb/announcing-ai-alignment-workshop-at-the-alife-2023",0,"",""],["Announcing AI Alignment workshop at the ALIFE 2023 conference","rorygreig","2023","blog","LessWrong","www.lesswrong.com/posts/LQAQe9Bkgb3oqFCaJ/announcing-ai-alignment-workshop-at-the-alife-2023",0,"",""],["Constructive Discussion and Thinking Methodology for Severe Situations including Existential Risks","Aino","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wbi3y8PsswpYvn5YN/constructive-discussion-and-thinking-methodology-for-severe",0,"","policy"],["Continuous Adversarial Quality Assurance: Extending RLHF and Constitutional AI","Benaya Koren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QGaioedKBJE39YJeD/continuous-adversarial-quality-assurance-extending-rlhf-and",0,"","rlhf constitutional-ai assurance"],["Minetester: A fully open RL environment built on Minetest","Curtis Huebner and 4 others","2023","blog","blog.eleuther.ai","blog.eleuther.ai/minetester-intro/",0,"",""],["Really Strong Features Found in Residual Stream","Logan Riggs","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Q76CpqHeEMykKpFdB/really-strong-features-found-in-residual-stream",0,"","interpretability"],["Seven Strategies for Tackling the Hard Part of the Alignment Problem","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/amBsmfFK4NFDtkHiT/seven-strategies-for-tackling-the-hard-part-of-the-alignment",0,"",""],["Views on when AGI comes and on strategy to reduce existential risk","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sTDfraZab47KiRMmT/views-on-when-agi-comes-and-on-strategy-to-reduce",0,"","forecasting"],["Views on when AGI comes and on strategy to reduce existential risk","TsviBT","2023","blog","LessWrong","www.lesswrong.com/posts/sTDfraZab47KiRMmT/views-on-when-agi-comes-and-on-strategy-to-reduce",0,"","forecasting"],["What Does LessWrong/EA Think of Human Intelligence Augmentation as of mid-2023?","marc/er","2023","blog","LessWrong","www.lesswrong.com/posts/ewitKJEwvttzk6zMi/what-does-lesswrong-ea-think-of-human-intelligence",0,"","scalable-oversight"],["What is everyone doing in AI governance","Igor Ivanov","2023","blog","LessWrong","www.lesswrong.com/posts/6gJMwXuWd3oskaLCo/what-is-everyone-doing-in-ai-governance",0,"","governance"],["Will the vast majority of technological progress happen in the longterm future?","Vasco Grilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/apZwBKDope6xqP3CT/will-the-vast-majority-of-technological-progress-happen-in",0,"",""],["Announcing the Existential InfoSec Forum","calebp and Wim van der Schoot","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dqpR2E4Bw9KEEaWoK/announcing-the-existential-infosec-forum",0,"",""],["Apparently, of the 195 Million the DoD allocated in University Research Funding Awards in 2022, more than half of them concerned AI or compute hardware research","mako yass","2023","blog","LessWrong","www.lesswrong.com/posts/rEDjo94iPvXWkkt4L/apparently-of-the-195-million-the-dod-allocated-in",0,"","governance"],["Internal independent review for language model agent alignment","Seth Herd","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Q7XWGqL4HjjRmhEyG/internal-independent-review-for-language-model-agent",0,"","agents chain-of-thought-faithfulness"],["Next week I'm interviewing tech policy expert Teddy Collins who has worked in the White House, DeepMind and CSET. What should I ask him?","Robert_Wiblin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/S29vagWmrAcFzzpm9/next-week-i-m-interviewing-tech-policy-expert-teddy-collins",0,"","policy"],["A Defense of Work on Mathematical AI Safety","Davidmanheim","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NKMxC2nA47uuhFm8x/a-defense-of-work-on-mathematical-ai-safety",0,"",""],["Concrete open problems in mechanistic interpretability: a technical overview","Neel Nanda","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EMfLZXvwiEioPWPga/concrete-open-problems-in-mechanistic-interpretability-a",0,"","interpretability mechanistic-interpretability"],["Empirical Evidence Against \"The Longest Training Run\"","NickGabs","2023","blog","LessWrong","www.lesswrong.com/posts/DaeHpWxvht43zaaje/empirical-evidence-against-the-longest-training-run",0,"","governance forecasting"],["Frontier AI regulation: Managing emerging risks to public safety","Markus Anderljung and 5 others","2023","blog","openai.com","openai.com/research/frontier-ai-regulation",0,"","governance"],["Jesse Hoogland on Developmental Interpretability and Singular Learning Theory","Michaël Trazzi","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LmHJPWRMEvmCWe382/jesse-hoogland-on-developmental-interpretability-and",0,"","interpretability"],["Localizing goal misgeneralization in a maze-solving policy network","jan betley","2023","blog","LessWrong","www.lesswrong.com/posts/vY9oE39tBupZLAyoC/localizing-goal-misgeneralization-in-a-maze-solving-policy",0,"","interpretability policy"],["The Unknowable Catastrophe","Aino","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qCfXKWYjJwqyNqbpd/the-unknowable-catastrophe",0,"",""],["€200k in European AI & Society Fund grants","Artūrs Kaņepājs","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eDNtcAyrNqaCenbos/eur200k-in-european-ai-and-society-fund-grants",0,"","policy"],["(tentatively) Found 600+ Monosemantic Features in a Small LM Using Sparse Autoencoders","Logan Riggs","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wqRqb7h6ZC48iDgfK/tentatively-found-600-monosemantic-features-in-a-small-lm",0,"","interpretability mechanistic-interpretability"],["[Linkpost] Introducing Superalignment","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hna4aoMwr6Qx9rHBs/linkpost-introducing-superalignment",0,"","automated-alignment-research"],["AISN #13: An interdisciplinary perspective on AI proxy failures, new competitors to ChatGPT, and prompting language models to misbehave","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iLnkLJE4JFPHHrASe/aisn-13-an-interdisciplinary-perspective-on-ai-proxy",0,"",""],["Announcing Manifund Regrants","Austin and Rachel Weinberg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RMXctNAksBgXgoszY/announcing-manifund-regrants",0,"",""],["Exploring Functional Decision Theory (FDT) and a modified version (ModFDT)","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/DMtzwPuFQtDmPEppF/exploring-functional-decision-theory-fdt-and-a-modified",0,"","theory"],["Know a grad student studying AI's economic impacts?","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bnde7LPfAazbEQnkZ/know-a-grad-student-studying-ai-s-economic-impacts",0,"","governance policy"],["OpenAI is starting a new \"Superintelligence alignment\" team and they're hiring","alejandro","2023","blog","EA Forum","forum.effectivealtruism.org/posts/idX6s3tTwRCXp94wY/openai-is-starting-a-new-superintelligence-alignment-team",0,"",""],["Optimized for Something other than Winning or: How Cricket Resists Moloch and Goodhart's Law","A.H.","2023","blog","LessWrong","www.lesswrong.com/posts/AZ4Hx7br9v5m5KbNm/optimized-for-something-other-than-winning-or-how-cricket",0,"","goodharts-law"],["Washington Post article about EA university groups","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cwZ8EvhWKTGbMKqzw/washington-post-article-about-ea-university-groups",0,"",""],["What did AI Safety’s specific funding of AGI R&D labs lead to?","Remmelt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XZDSBSpr897eR6cBW/what-did-ai-safety-s-specific-funding-of-agi-r-and-d-labs",0,"",""],["[linkpost] Ten Levels of AI Alignment Difficulty","SammyDMartin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cp4aQuFfH5PwAJmdz/linkpost-ten-levels-of-ai-alignment-difficulty",0,"",""],["AI labs' statements on governance","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/iFrefmWAct3wYG7vQ/ai-labs-statements-on-governance",0,"","governance"],["Animal Weapons: Lessons for Humans in the Age of X-Risk","Damin Curtis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zimRZudzaewHJu4NA/animal-weapons-lessons-for-humans-in-the-age-of-x-risk-2",0,"",""],["The Neural Net Tank Urban Legend","Gwern Branwen","2023","blog","gwern.net","www.gwern.net/Tanks.page",0,"",""],["Ways I Expect AI Regulation To Increase Extinction Risk","1a3orn","2023","blog","LessWrong","www.lesswrong.com/posts/6untaSPpsocmkS7Z3/ways-i-expect-ai-regulation-to-increase-extinction-risk",0,"","governance"],["[Job] Managing Director at the Cooperative AI Foundation ($5000 Referral Bonus)","Lewis Hammond","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Xyu5NJ5tyJM8xFcJJ/job-managing-director-at-the-cooperative-ai-foundation",0,"",""],["Douglas Hoftstadter concerned about AI xrisk","Eli Rose","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3sJbwpGbAu5tpGkqD/douglas-hoftstadter-concerned-about-ai-xrisk",0,"",""],["My Alignment Timeline","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/RTjE6KN2WGepGL6m9/my-alignment-timeline",0,"","forecasting"],["Ten Levels of AI Alignment Difficulty","Sammy Martin","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EjgfreeibTXRx9Ham/ten-levels-of-ai-alignment-difficulty",0,"","alignment-faking deception"],["(Intro/1) - My Understandings of Mechanistic Interpretability Notebook","Yadav","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Cmb6Wpbo5NyjzD2zo/intro-1-my-understandings-of-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability"],["Apply to fall policy internships (we can help)","Elika and Vaidehi Agarwalla","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pdMjPuddtHeLSBDiF/apply-to-fall-policy-internships-we-can-help",0,"","governance policy"],["How Smart Are Humans?","Joar Skalse","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/7uexzZka8YtQzMwPf/how-smart-are-humans",0,"","forecasting"],["Quantitative cruxes in Alignment","Martín Soto","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ryCfHod3eFhkYipW9/quantitative-cruxes-in-alignment",0,"","forecasting"],["Sources of evidence in Alignment","Martín Soto","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9sjEFuGbb8WFZE9ER/sources-of-evidence-in-alignment",0,"","forecasting"],["Using (Uninterpretable) LLMs to Generate Interpretable AI Code","Joar Skalse","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sCJDstZrpCB8dQveA/using-uninterpretable-llms-to-generate-interpretable-ai-code",0,"","interpretability"],["VC Theory Overview","Joar Skalse","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uEwECj53prjKLcBC5/vc-theory-overview",0,"",""],["Elements of Computational Philosophy, Vol. I: Truth","Paul Bricman and Tom Feeney","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/k8xreAj2nXGra22H7/elements-of-computational-philosophy-vol-i-truth",0,"",""],["Agency from a causal perspective","tom4everitt and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qi77Tu3ehdacAbBBe/agency-from-a-causal-perspective",0,"",""],["Foom Liability","PeterMcCluskey","2023","blog","LessWrong","www.lesswrong.com/posts/yEQuEsWPQAaXzhdxz/foom-liability",0,"","governance"],["George Hotz on AI safety: ~\"centralized power is bad\"","Chipmonk","2023","blog","LessWrong","www.lesswrong.com/posts/5dMavpaByQaurxkYq/george-hotz-on-ai-safety-centralized-power-is-bad",0,"","power-seeking"],["Inherently Interpretable Architectures","Robert Kralisch and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/FBcK7dEBSTgsHLwET/inherently-interpretable-architectures",0,"","interpretability"],["Introducing EffiSciences’ AI Safety Unit","WCargo and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DkDy2hvkwbQ54GM9u/introducing-effisciences-ai-safety-unit-1",0,"",""],["Introducing EffiSciences’ AI Safety Unit","WCargo and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/DkDy2hvkwbQ54GM9u/introducing-effisciences-ai-safety-unit-1",0,"",""],["Introduction","Robert Kralisch and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/xfGKpLevfogGefDfx/introduction-9",0,"",""],["Little attention seems to be on discouraging hardware progress","RussellThor","2023","blog","LessWrong","www.lesswrong.com/posts/HziboapdPFqaF5gaW/little-attention-seems-to-be-on-discouraging-hardware",0,"","governance"],["On Agent Foundations","Robert Kralisch","2023","blog","LessWrong","www.lesswrong.com/posts/7ZkQ5wGouiBeRMSdF/on-agent-foundations",0,"","agents theory"],["Positive Attractors","Robert Kralisch and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/9aN7hrCFhoQshqQz2/positive-attractors",0,"",""],["Three camps in AI x-risk discussions: My personal very oversimplified overview","Aryeh Englander","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kgPJMJxahHeQMFR5d/three-camps-in-ai-x-risk-discussions-my-personal-very",0,"",""],["AI Safety without Alignment: How humans can WIN against AI","vicchain","2023","blog","LessWrong","www.lesswrong.com/posts/3wMAppMNRQvAbhwcj/ai-safety-without-alignment-how-humans-can-win-against-ai",0,"",""],["Anthropically Blind: the anthropic shadow is reflectively inconsistent","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/LGHuaLiq3F5NHQXXF/anthropically-blind-the-anthropic-shadow-is-reflectively",0,"","theory"],["Biosafety Regulations (BMBL) and their relevance for AI","stepanlos","2023","blog","EA Forum","forum.effectivealtruism.org/posts/g38CkMbFzKBtdzFXY/biosafety-regulations-bmbl-and-their-relevance-for-ai",0,"","governance policy"],["Biosafety Regulations (BMBL) and their relevance for AI","Štěpán Los","2023","blog","LessWrong","www.lesswrong.com/posts/tTRWgmtPetrxiEtSs/biosafety-regulations-bmbl-and-their-relevance-for-ai",0,"","governance"],["Challenge proposal: smallest possible self-hardening backdoor for RLHF","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/QPqFJ8oEEuzxqsatw/challenge-proposal-smallest-possible-self-hardening-backdoor-1",0,"","rlhf evals"],["Cheat sheet of AI X-risk","amaury lorin","2023","blog","LessWrong","www.lesswrong.com/posts/nCeyBbhtJhToBFmrL/cheat-sheet-of-ai-x-risk",0,"","forecasting"],["Updates from Campaign for AI Safety","Jolyn Khoo and Nik Samoylov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YmEMGbnwuWeGLbnzv/updates-from-campaign-for-ai-safety",0,"",""],["[Link Post] Interesting shallow round-up of reasons to be skeptical that transformative AI or explosive economic growth are coming soon","Dr. David Mathers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aXGDHeyhaep5sLzuG/link-post-interesting-shallow-round-up-of-reasons-to-be",0,"",""],["A \"weak\" AGI may attempt an unlikely-to-succeed takeover","RobertM","2023","blog","LessWrong","www.lesswrong.com/posts/B5CNPqYL7XcHzgzHc/a-weak-agi-may-attempt-an-unlikely-to-succeed-takeover",0,"","alignment-faking deception"],["AGI x Animal Welfare: A High-EV Outreach Opportunity?","simeon_c","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZnPLPFC49nJym7y8g/agi-x-animal-welfare-a-high-ev-outreach-opportunity",0,"",""],["AI & Drug Discovery - Security and Risks","Girving","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Ya8D8jfcg2qaYEYim/ai-and-drug-discovery-security-and-risks",0,"","policy"],["AI Incident Sharing - Best practices from other fields and a comprehensive list of existing platforms","stepanlos","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dikcpP32Q3cg6tvdA/ai-incident-sharing-best-practices-from-other-fields-and-a",0,"","policy"],["AI Incident Sharing - Best practices from other fields and a comprehensive list of existing platforms","Štěpán Los","2023","blog","LessWrong","www.lesswrong.com/posts/sAt6zfeatgFiikAkE/ai-incident-sharing-best-practices-from-other-fields-and-a",0,"","governance"],["Brief summary of ai-plans.com","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/iPESsoBuXdzvaA675/brief-summary-of-ai-plans-com",0,"",""],["Carl Shulman on The Lunar Society (7 hour, two-part podcast)","ESRogs","2023","blog","LessWrong","www.lesswrong.com/posts/Ghrdnc26ftJrxD49z/carl-shulman-on-the-lunar-society-7-hour-two-part-podcast",0,"","forecasting"],["My research agenda in agent foundations","Alex_Altair","2023","blog","LessWrong","www.lesswrong.com/posts/7nDvJiikgiawHAp6z/my-research-agenda-in-agent-foundations",0,"","agents theory"],["Towards Measuring the Representation of Subjective Global Opinions in Language Models","Esin Durmus  Karina Nguyen  Thomas I. Liao  Nicholas Schiefer","2023","paper","arXiv preprint","arxiv.org/abs/2306.16388",0,"","constitutional-ai evals"],["When do \"brains beat brawn\" in Chess? An experiment","titotal","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/odtMt7zbMuuyavaZB/when-do-brains-beat-brawn-in-chess-an-experiment",0,"",""],["AISC team report: Soft-optimization, Bayes and Goodhart","Simon Fischer and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/XXrGhqSNZjcG2nNiy/aisc-team-report-soft-optimization-bayes-and-goodhart",0,"","goodharts-law"],["AISN #12: Policy Proposals from NTIA’s Request for Comment and Reconsidering Instrumental Convergence","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MqhzeTpemwHcxrwzd/aisn-12-policy-proposals-from-ntia-s-request-for-comment-and",0,"","instrumental-convergence policy"],["An overview of the points system","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/hcTiw9xKNZAi7qcy6/an-overview-of-the-points-system",0,"",""],["Catastrophic Risks from AI #5: Rogue AIs","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nJEJAcS6Bs4BJbkZb/catastrophic-risks-from-ai-5-rogue-ais",0,"",""],["Catastrophic Risks from AI #6: Discussion and FAQ","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ydtdwWSCCihms5Jeo/catastrophic-risks-from-ai-6-discussion-and-faq",0,"",""],["ML4G Germany - AI Alignment Camp","Evander H.","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wncsWoJnpEid3dJy8/ml4g-germany-ai-alignment-camp-1",0,"",""],["\"Safety Culture for AI\" is important, but isn't going to be easy","Davidmanheim","2023","blog","LessWrong","www.lesswrong.com/posts/iFLNKgZceYyTdwsGz/safety-culture-for-ai-is-important-but-isn-t-going-to-be",0,"","governance"],["AI Safety Field Building vs. EA CB","kuhanj","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YpADfSeSccsEkaetk/ai-safety-field-building-vs-ea-cb",0,"",""],["Catastrophic Risks from AI #4: Organizational Risks","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wpsGprQCRffRKG92v/catastrophic-risks-from-ai-4-organizational-risks",0,"",""],["Deceptive AI vs. shifting instrumental incentives","Aryeh Englander","2023","blog","LessWrong","www.lesswrong.com/posts/rX4AdyrtjYrEX9xnn/deceptive-ai-vs-shifting-instrumental-incentives",0,"","alignment-faking deception"],["Import AI 333: Synthetic data makes models stupid; chatGPT eats MTurk. Inflection shows off a large language model","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-333-synthetic-data-makes",0,"",""],["Let’s set new AI safety actors up for success","michel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hHhGyxshSJg43yfqH/let-s-set-new-ai-safety-actors-up-for-success",0,"","governance"],["Looking for Canadian summer co-op position in AI Governance","tcelferact","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qJmadiMjhLbYgqjXj/looking-for-canadian-summer-co-op-position-in-ai-governance",0,"","governance"],["The EU AI Act: A Simple Explanation - A Stanford Study Reveals the gaps of ChatGPT and 9 more","Sparkvibe","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pdKDjEhmANeD8BQhb/the-eu-ai-act-a-simple-explanation-a-stanford-study-reveals",0,"","policy"],["The fraught voyage of aligned novelty","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ASZco85chGouu2LKk/the-fraught-voyage-of-aligned-novelty",0,"",""],["Where on the continuum of pure EA to pure AIS should you be? (Uni Group Organizers Focus)","jessica_mccurdy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kjYx6BvTKKJg3xQie/where-on-the-continuum-of-pure-ea-to-pure-ais-should-you-be",0,"",""],["Did Bengio and Tegmark lose a debate about AI x-risk against LeCun and Mitchell?","Karl von Wendt","2023","blog","LessWrong","www.lesswrong.com/posts/CA7iLZHNT5xbLK59Y/did-bengio-and-tegmark-lose-a-debate-about-ai-x-risk-against",0,"",""],["Map of maps of interesting fields","Max Görlitz","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dAuaHKnH6CsaH8ecg/map-of-maps-of-interesting-fields",0,"",""],["Would a super-intelligent AI necessarily support its own existence?","Porque?","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nNCdhMenNYRcRFWab/would-a-super-intelligent-ai-necessarily-support-its-own",0,"",""],["Democratic AI Constitution: Round-Robin Debate and Synthesis","scottviteri","2023","blog","LessWrong","www.lesswrong.com/posts/habXKpXKaSK2F7vDP/democratic-ai-constitution-round-robin-debate-and-synthesis",0,"","governance"],["DSLT 4. Phase Transitions in Neural Networks","Liam Carroll","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aKBAYN5LpaQMrPqMj/dslt-4-phase-transitions-in-neural-networks",0,"",""],["Announcing the AIPolicyIdeas.com Database","abiolvera","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cJLsd2TYxv8KCzHvg/announcing-the-aipolicyideas-com-database",0,"","policy"],["Catastrophic Risks from AI #3: AI Race","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4sEK5mtDYWJo2gHJn/catastrophic-risks-from-ai-3-ai-race",0,"",""],["On the compute governance era and what has to come after (Lennart Heim on The 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/76iiCpwiJNKCGZd9o/on-the-compute-governance-era-and-what-has-to-come-after",0,"","governance compute-governance"],["OpenAI's grant program for democratic process for deciding what rules AI systems should follow","Ronen Bar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/azYjfNsMokoJhMdc4/openai-s-grant-program-for-democratic-process-for-deciding",0,"","governance policy"],["Slaying the Hydra: toward a new game board for AI","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/9xaW2yQRpyjp23ikg/slaying-the-hydra-toward-a-new-game-board-for-ai",0,"","governance forecasting"],["Thoughts about AI safety field-building in LMIC","Renan Araujo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GmDmE2pxjTjHMHNK8/thoughts-about-ai-safety-field-building-in-lmic",0,"",""],["What should I ask Ezra Klein about AI policy proposals?","Robert_Wiblin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CQ6gNxRhwxhnw2bKF/what-should-i-ask-ezra-klein-about-ai-policy-proposals",0,"","governance policy"],["Announcing the EA Project Ideas Database","Joe Rogero","2023","blog","EA Forum","forum.effectivealtruism.org/posts/S4hpXjJ5cHvn6LkLu/announcing-the-ea-project-ideas-database",0,"",""],["Catastrophic Risks from AI #1: Summary","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bvdbx6tW9yxfxAJxe/catastrophic-risks-from-ai-1-summary",0,"",""],["Catastrophic Risks from AI #2: Malicious Use","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MtDmnSpPHDvLr7CdM/catastrophic-risks-from-ai-2-malicious-use",0,"",""],["RP’s AI Governance & Strategy team - June 2023 interim overview","MichaelA","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BSmMok4r5ocnD5dqT/rp-s-ai-governance-and-strategy-team-june-2023-interim-1",0,"","governance policy compute-governance forecasting"],["The Hubinger lectures on AGI safety: an introductory lecture series","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jotZXEixzToEHnrfr/the-hubinger-lectures-on-agi-safety-an-introductory-lecture",0,"",""],["US public perception of CAIS statement and the risk of extinction","Jamie Elsey and David_Moss","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Rg7h7G3KTvaYEtL55/us-public-perception-of-cais-statement-and-the-risk-of",0,"","policy"],["Why Not Subagents?","johnswentworth and David Lorell","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bzmLC3J8PsknwRZbr/why-not-subagents",0,"","agents"],["Yip Fai Tse on animal welfare & AI safety and long termism","Karthik Palakodeti and Fai","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gxAXKRTzdEqiRbkrr/yip-fai-tse-on-animal-welfare-and-ai-safety-and-long-termism",0,"",""],["20 concrete projects for reducing existential risk","Buhl and Jam Kraprayoon","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AJwuMw7ddcKQNFLcR/20-concrete-projects-for-reducing-existential-risk",0,"","evals"],["A way to make solving alignment 10.000 times easier. The shorter case for a massive open source simbox project.","AlexFromSafeTransition","2023","blog","LessWrong","www.lesswrong.com/posts/EKN8Zv4hZY3hMywKz/a-way-to-make-solving-alignment-10-000-times-easier-the",0,"","alignment-faking deception"],["An Overview of Catastrophic AI Risks","Dan Hendrycks and 2 others","2023","paper","arXiv preprint","arxiv.org/abs/2306.12001",0,"","agents policy"],["EU AI Act passed Plenary vote, and X-risk was a main topic","Ariel G.","2023","blog","LessWrong","www.lesswrong.com/posts/L5pWY8gEGhsHiWcsG/eu-ai-act-passed-plenary-vote-and-x-risk-was-a-main-topic",0,"","governance"],["Join the Virtual AI Safety Unconference (VAISU)!","Nguyên and Linda Linsefors","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zmxckAoi9DjDyHsyA/join-the-virtual-ai-safety-unconference-vaisu",0,"","governance"],["Upcoming speaker series on emerging tech, national security & US policy careers","kuhanj","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZZe5aFGKeZATYGGMD/upcoming-speaker-series-on-emerging-tech-national-security",0,"","policy"],["Using Claude to convert dialog transcripts into great posts?","mako yass","2023","blog","LessWrong","www.lesswrong.com/posts/4SNZmgm7iNuK25cai/using-claude-to-convert-dialog-transcripts-into-great-posts",0,"",""],["A Friendly Face (Another Failure Story)","Karl von Wendt and 5 others","2023","blog","LessWrong","www.lesswrong.com/posts/iRFxvNeLbHNRCzA2S/a-friendly-face-another-failure-story",0,"",""],["Ban development of unpredictable powerful models?","TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8CvkNa6FKSrK4Nj83/ban-development-of-unpredictable-powerful-models",0,"","governance"],["Causality: A Brief Introduction","tom4everitt and 6 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9ag5JGBnMsayBidwh/causality-a-brief-introduction",0,"",""],["Corrigibility test #1: Shutdown activations in a Virus Research Lab","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/mksPEJhR78SyDiyGz/corrigibility-test-1-shutdown-activations-in-a-virus",0,"",""],["DSLT 3. Neural Networks are Singular","Liam Carroll","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tZwaGp5wMQqKh3krz/dslt-3-neural-networks-are-singular",0,"",""],["Lightning Post: Things people in AI Safety should stop talking about","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/TpExcpmeHhhfNtXoh/lightning-post-things-people-in-ai-safety-should-stop",0,"",""],["LPP Summer Research Fellowship in Law & AI 2023: Applications Open","Legal Priorities Project","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QXywXmka8pACPuiHq/lpp-summer-research-fellowship-in-law-and-ai-2023",0,"","policy"],["Simulating Shutdown Code Activations in an AI Virus Lab","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/w58znDmKpfqvYoGdY/simulating-shutdown-code-activations-in-an-ai-virus-lab",0,"",""],["Summary of the AI Bill of Rights and Policy Implications","Tristan Williams","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RLgjq9GFz9iHKC4ZZ/summary-of-the-ai-bill-of-rights-and-policy-implications",0,"","governance policy"],["A Multidisciplinary Approach to Alignment (MATA) and Archetypal Transfer Learning (ATL)","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/zQ4dX8Jk4uExukxqB/a-multidisciplinary-approach-to-alignment-mata-and",0,"",""],["Experiments in Evaluating Steering Vectors","Gytis Daujotas","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GGttHsdWHjh94cX5G/experiments-in-evaluating-steering-vectors",0,"","evals"],["Mode collapse in RL may be fueled by the update equation","TurnTrout and MichaelEinhorn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/A7RgYuYH4HywNeYWD/mode-collapse-in-rl-may-be-fueled-by-the-update-equation",0,"","rlhf"],["New reference standard on LLM Application security started by OWASP","QuantumForest","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mdg8gL59LiiZmaGCw/new-reference-standard-on-llm-application-security-started",0,"",""],["Principles for AI Welfare Research","jeffsebo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SZJBE3fuk2majqwJQ/principles-for-ai-welfare-research",0,"","governance"],["Provisionality","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5ie8Mfwq5tEBCBDEC/provisionality",0,"",""],["The Multidisciplinary Approach to Alignment (MATA) and Archetypal Transfer Learning (ATL)","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PhG6vQahtDhsmQ4BE/the-multidisciplinary-approach-to-alignment-mata-and",0,"",""],["DSLT 2. Why Neural Networks obey Occam's Razor","Liam Carroll","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CZHwwDd7t9aYra5HN/dslt-2-why-neural-networks-obey-occam-s-razor",0,"",""],["My lab's small AI safety agenda","Jobst Heitzig (vodle.it)","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZWjDkENuFohPShTyc/my-lab-s-small-ai-safety-agenda",0,"",""],["UK Foundation Model Task Force - Expression of Interest","ojorgensen","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qTAE9GZ9kAKnstvJ4/uk-foundation-model-task-force-expression-of-interest",0,"","policy"],["A summary of current work in AI governance","constructive","2023","blog","LessWrong","www.lesswrong.com/posts/GEJtDHMfuW4vZ5msG/a-summary-of-current-work-in-ai-governance",0,"","governance"],["Partial Simulation Extrapolation: A Proposal for Building Safer Simulators","marc/er","2023","blog","LessWrong","www.lesswrong.com/posts/3pCdCJxQRKffY2NTu/partial-simulation-extrapolation-a-proposal-for-building",0,"",""],["The AI governance gaps in developing countries","nguyên","2023","blog","LessWrong","www.lesswrong.com/posts/cM7sR7seBRwtxctGY/the-ai-governance-gaps-in-developing-countries",0,"","governance"],["[Replication] Conjecture's Sparse Coding in Small Transformers","Hoagy and Logan Riggs","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vBcsAw4rvLsri3JAj/replication-conjecture-s-sparse-coding-in-small-transformers",0,"",""],["Conjecture: A standing offer for public debates on AI","Andrea_Miotti","2023","blog","LessWrong","www.lesswrong.com/posts/q5kio5Tz4C3bDJ8N6/conjecture-a-standing-offer-for-public-debates-on-ai",0,"",""],["Critiques of non-existent AI safety labs: Yours","Anneal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PJLx7CwB4mtaDgmFc/critiques-of-non-existent-ai-safety-labs-yours",0,"",""],["Does anyone's full-time job include reading and understanding all the most-promising formal AI alignment work?","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/m25oFp3DzmJYH77df/does-anyone-s-full-time-job-include-reading-and",0,"",""],["DSLT 0. Distilling Singular Learning Theory","Liam Carroll","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/xRWsfGfvDAjRWXcnG/dslt-0-distilling-singular-learning-theory",0,"",""],["DSLT 1. The RLCT Measures the Effective Dimension of Neural Networks","Liam Carroll","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4eZtmwaqhAgdJQDEg/dslt-1-the-rlct-measures-the-effective-dimension-of-neural",0,"",""],["LLMs Sometimes Generate Purely Negatively-Reinforced Text","Fabien Roger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sbGau4QBwToYWEg4k/llms-sometimes-generate-purely-negatively-reinforced-text",0,"",""],["Safety evaluations and standards for AI | Beth Barnes | EAG Bay Area 23","Beth Barnes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/49rzRKh2ZYH2QjPkg/safety-evaluations-and-standards-for-ai-or-beth-barnes-or",0,"","evals"],["Scaffolded LLMs: Less Obvious Concerns","Stephen Fowler","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mAwxebLw3nYbDivmt/scaffolded-llms-less-obvious-concerns",0,"","chain-of-thought-faithfulness"],["Updates from Campaign for AI Safety","Jolyn Khoo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/r8q7mzfxqr8fxrEfH/updates-from-campaign-for-ai-safety-2",0,"",""],["Updating Drexler's CAIS model","Matthew Barnett","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/a5NxvzFGddj2e8uXQ/updating-drexler-s-cais-model",0,"",""],["What would it look like for AIS to no longer be neglected?","Rockwell","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nhenCNq7s3zaXpQ8c/what-would-it-look-like-for-ais-to-no-longer-be-neglected",0,"",""],["Aligned Objectives Prize Competition","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/8wWb9zhwk8qH4cSK7/aligned-objectives-prize-competition",0,"",""],["AXRP Episode 22 - Shard Theory with Quintin Pope","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4rmvMThJYNcCptAya/axrp-episode-22-shard-theory-with-quintin-pope",0,"",""],["Brief thoughts on Data, Reporting, and Response for AI Risk Mitigation","Davidmanheim","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TfkufmAQG9PjmE6jo/brief-thoughts-on-data-reporting-and-response-for-ai-risk",0,"",""],["EU AI Act passed vote, and x-risk was a main topic","Ariel G.","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ctPrrzFnXGyWrmK3w/eu-ai-act-passed-vote-and-x-risk-was-a-main-topic",0,"","policy"],["human intelligence may be alignment-limited","bhauth","2023","blog","LessWrong","www.lesswrong.com/posts/zGoDXgLFetEdvLdFH/human-intelligence-may-be-alignment-limited",0,"","instrumental-convergence"],["Inverse Scaling: When Bigger Isn’t Better","\\name","2023","paper","arXiv preprint","arxiv.org/abs/2306.09479",0,"","training-data scaling-laws"],["PhD student and postdoc positions philosophy of AI in Erlangen (Germany)","LeonardDung","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Z33e5G6CLCkvg2t7F/phd-student-and-postdoc-positions-philosophy-of-ai-in",0,"",""],["Report: Artificial Intelligence Risk Management in Spain","JorgeTorresC and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wRQx4tBtqpqF4QE3h/report-artificial-intelligence-risk-management-in-spain",0,"","evals governance policy"],["UN Secretary-General recognises existential threat from AI","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wxxoRHmisojF6Y2qD/un-secretary-general-recognises-existential-threat-from-ai",0,"","governance policy"],["Why \"AI alignment\" would better be renamed into \"Artificial Intention research\"","chaosmage","2023","blog","LessWrong","www.lesswrong.com/posts/MksJyLgJ8JQFexiWi/why-ai-alignment-would-better-be-renamed-into-artificial",0,"",""],["a short chat about realityfluid","Tamsin Leake","2023","blog","carado.moe","carado.moe/short-chat-realityfluid.html",0,"",""],["AI Safety Strategy - A new organization for better timelines","Prometheus","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XQJNnFryp8ZFqk6td/ai-safety-strategy-a-new-organization-for-better-timelines",0,"","governance forecasting"],["Anthropic | Charting a Path to AI Accountability","Gabriel Mukobi","2023","blog","LessWrong","www.lesswrong.com/posts/L9xDcTBwNyciyNbb2/anthropic-or-charting-a-path-to-ai-accountability",0,"","governance"],["Demystifying Born's rule","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/7nJt3hkQg9uSEta6M/demystifying-born-s-rule",0,"","theory"],["Instrumental Convergence? [Draft]","J. Dmitri Gallow","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/w8PNjCS8ZsQuqYWhD/instrumental-convergence-draft",0,"","instrumental-convergence"],["Linkpost: Dwarkesh Patel interviewing Carl Shulman","Stefan_Schubert","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iKWeQ7jhsZ8FrRDto/linkpost-dwarkesh-patel-interviewing-carl-shulman",0,"",""],["<$750k grants for General Purpose AI Assurance/Safety Research","Phosphorous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vz8wia5x8pk2fNLZk/less-than-usd750k-grants-for-general-purpose-ai-assurance",0,"","assurance"],["Aptitudes for AI governance work","Sam Clarke","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ozSBaNLysue9MmFqs/aptitudes-for-ai-governance-work",0,"","governance"],["Epoch and FRI Mentorship Program Summer 2023","merilalama","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QiCZoxjjvPpd8qfWb/epoch-and-fri-mentorship-program-summer-2023-1",0,"","forecasting"],["Introducing The Long Game Project: Improving Decision-Making Through Tabletop Exercises and Simulated Experience","Dan Stuart","2023","blog","LessWrong","www.lesswrong.com/posts/vcSQzNJDGKLG8fgTE/introducing-the-long-game-project-improving-decision-making",0,"","theory"],["MetaAI: less is less for alignment.","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uyk5nn93HxJMsio98/metaai-less-is-less-for-alignment-1",0,"","rlhf alignment-faking deception"],["Raising the voices that actually count","Kim Holder","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oyf3qbXSh8FJGicof/raising-the-voices-that-actually-count",0,"",""],["Some talent needs in AI governance","Sam Clarke","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gsPmsdXWFmkwezc5L/some-talent-needs-in-ai-governance",0,"","governance"],["TASRA: A Taxonomy and Analysis of Societal-Scale Risks from AI","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zKkZanEQc4AZBEKx9/tasra-a-taxonomy-and-analysis-of-societal-scale-risks-from",0,"",""],["There is only one goal or drive - only self-perpetuation counts","freest one","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DTPR8agC36kojCq9j/there-is-only-one-goal-or-drive-only-self-perpetuation",0,"","instrumental-convergence"],["Tony Blair Institute AI Safety Work","TomWestgarth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EBZggasznbotKrpLW/tony-blair-institute-ai-safety-work",0,"",""],["What's the exact way you predict probability of AI extinction?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nZxcd5LGBNfFC9ejY/what-s-the-exact-way-you-predict-probability-of-ai",0,"","forecasting"],["A Manifold Market \"Leaked\" the AI Extinction Statement and CAIS Wanted it Deleted","David Chee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zJxRcb4TCCouh3eaa/a-manifold-market-leaked-the-ai-extinction-statement-and",0,"","forecasting"],["ARC is hiring theoretical researchers","paulfchristiano and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/d99ikjqdxpMiAFnch/arc-is-hiring-theoretical-researchers-1",0,"",""],["ARC is hiring theoretical researchers","Jacob_Hilton and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jwCym3zqbztA8qRZ4/arc-is-hiring-theoretical-researchers",0,"",""],["Contingency: A Conceptual Tool from Evolutionary Biology for Alignment","clem_acs","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GAbEcLDQLtHZE78TY/contingency-a-conceptual-tool-from-evolutionary-biology-for-2",0,"","situational-awareness instrumental-convergence"],["Critiques of prominent AI safety labs: Conjecture","Omega","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gkfMLX4NWZdmpikto/critiques-of-prominent-ai-safety-labs-conjecture",0,"",""],["Critiques of prominent AI safety labs: Conjecture","Omega.","2023","blog","LessWrong","www.lesswrong.com/posts/9jvrQToSq3CYvoeHf/critiques-of-prominent-ai-safety-labs-conjecture",0,"",""],["Explicitness","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KuKaQEu7JjBNzcoj5/explicitness",0,"",""],["If you are too stressed, walk away from the front lines","Neil Warren","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8WNaM6nSJ3wXALKCv/if-you-are-too-stressed-walk-away-from-the-front-lines",0,"",""],["Import AI 332: Mini-AI; safety through evals; Facebook releases a RLHF dataset","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-332-mini-ai-safety-through",0,"","rlhf evals"],["Introduction to Towards Causal Foundations of Safe AGI","tom4everitt and 6 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3oQCY4he4zRzF6vQb/introduction-to-towards-causal-foundations-of-safe-agi",0,"",""],["On DeepMind and Trying to Fairly Hear Out Both AI Doomers and Doubters (Rohin Shah on The 80,000 Hours Podcast)","80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mBKLGm9GXzoet9Gaq/on-deepmind-and-trying-to-fairly-hear-out-both-ai-doomers",0,"",""],["TASRA: a Taxonomy and Analysis of Societal-Scale Risks from AI","Andrew Critch and Stuart Russell","2023","paper","arXiv preprint","arxiv.org/abs/2306.06924",0,"","policy"],["What can superintelligent ANI tell us about superintelligent AGI?","Ted Sanders","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NPHJBby6KjDC7iNYK/what-can-superintelligent-ani-tell-us-about-superintelligent",0,"",""],["Higher Dimension Cartesian Objects and Aligning ‘Tiling Simulators’","marc/er","2023","blog","LessWrong","www.lesswrong.com/posts/aS3rNSww3jwkeAHjT/higher-dimension-cartesian-objects-and-aligning-tiling",0,"","agents theory"],["Inference-Time Intervention: Eliciting Truthful Answers from a Language Model","likenneth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kuQfnotjkQA4Kkfou/inference-time-intervention-eliciting-truthful-answers-from",0,"","interpretability"],["Mitigating Ethical Concerns and Risks in the US Approach to Autonomous Weapons Systems through Effective Altruism","Vee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QEifHsCzHzuKtF82F/mitigating-ethical-concerns-and-risks-in-the-us-approach-to",0,"","forecasting"],["an Evangelion dialogue explaining the QACI alignment plan","Tamsin Leake","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/i9okkiKQ4rY8eawmT/an-evangelion-dialogue-explaining-the-qaci-alignment-plan",0,"","agents theory"],["Are we confident that superintelligent artificial intelligence disempowering humans would be bad?","Vasco Grilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Y5eQHtEB29nW6FfQE/are-we-confident-that-superintelligent-artificial",0,"",""],["formalizing the QACI alignment formal-goal","Tamsin Leake and JuliaHP","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MR5wJpE27ymE7M7iv/formalizing-the-qaci-alignment-formal-goal",0,"","agents theory"],["Goal-misgeneralization is ELK-hard","rokosbasilisk","2023","blog","LessWrong","www.lesswrong.com/posts/MWSCqzPrAbNrYoqWv/goal-misgeneralization-is-elk-hard",0,"","eliciting-latent-knowledge"],["Using Consensus Mechanisms as an approach to Alignment","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/2SCSpN7BRoGhhwsjg/using-consensus-mechanisms-as-an-approach-to-alignment",0,"","governance forecasting"],["What are brains?","Valentine","2023","blog","LessWrong","www.lesswrong.com/posts/aBDsJhkMnFcos3HKC/what-are-brains",0,"","theory"],["A plea for solutionism on AI safety","jasoncrawford","2023","blog","LessWrong","www.lesswrong.com/posts/ASMX9ss3J5G3GZdok/a-plea-for-solutionism-on-ai-safety",0,"",""],["AI Rights: In your view, what would be required for an AGI to gain rights and protections from the various Governments of the World?","Super AGI","2023","blog","LessWrong","www.lesswrong.com/posts/2gHEfJfuP5ToQMv2E/ai-rights-in-your-view-what-would-be-required-for-an-agi-to",0,"","evals governance forecasting"],["an Evangelion dialogue explaining the QACI alignment plan","Tamsin Leake","2023","blog","carado.moe","carado.moe/qaci-invention-dialogue.html",0,"",""],["Announcement: You can now listen to the “AI Safety Fundamentals” courses","peterhartree and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vxpqFFtrRsG9RLkqa/announcement-you-can-now-listen-to-the-ai-safety",0,"","policy"],["formalizing the QACI alignment formal-goal","Tamsin Leake","2023","blog","carado.moe","carado.moe/qaci-math.html",0,"",""],["How biosafety could inform AI standards","Olivia Jimenez","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Tb2gAaHLmigfFGksq/how-biosafety-could-inform-ai-standards",0,"","governance"],["How does AI progress affect other EA cause areas?","Luis Mota Freitas","2023","blog","EA Forum","forum.effectivealtruism.org/posts/J4cLuxvAwnKNQxwxj/how-does-ai-progress-affect-other-ea-cause-areas",0,"","forecasting"],["Improvement on MIRI's Corrigibility","WCargo and Charbel-Raphaël","2023","blog","LessWrong","www.lesswrong.com/posts/fNwDEHWFnHMtm8yH4/improvement-on-miri-s-corrigibility",0,"",""],["[Linkpost] Scaling laws for language encoding models in fMRI","Bogdan Ionut Cirstea","2023","blog","LessWrong","www.lesswrong.com/posts/iXbPe9EAxScuimsGh/linkpost-scaling-laws-for-language-encoding-models-in-fmri",0,"","scaling-laws"],["A comparison of causal scrubbing, causal abstractions, and related methods","Erik Jenner and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uLMWMeBG3ruoBRhMW/a-comparison-of-causal-scrubbing-causal-abstractions-and",0,"","interpretability"],["A potentially high impact differential technological development area","Noosphere89","2023","blog","LessWrong","www.lesswrong.com/posts/8SpbjkJREzp2H4dBB/a-potentially-high-impact-differential-technological",0,"","instrumental-convergence automated-alignment-research"],["A survey of concrete risks derived from Artificial Intelligence","Guillem Bas and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/K3XiFGMdAQXBGTFSH/a-survey-of-concrete-risks-derived-from-artificial",0,"","governance"],["Beware popular discussions of AI \"sentience\"","Dr. David Mathers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KBMSJj63nZfsji2wS/beware-popular-discussions-of-ai-sentience",0,"",""],["Takeaways from the Mechanistic Interpretability Challenges","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EjsA2M8p8ERyFHLLY/takeaways-from-the-mechanistic-interpretability-challenges",0,"","interpretability mechanistic-interpretability"],["Transformative AI is a process","meijer1973","2023","blog","LessWrong","www.lesswrong.com/posts/PQaC6pmnPxF8DgpbJ/transformative-ai-is-a-process",0,"","forecasting"],["UK government to host first global summit on AI Safety","DavidNash","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Y2xbKLjEmL6dCd2Z6/uk-government-to-host-first-global-summit-on-ai-safety",0,"","governance"],["Wild Animal Welfare Scenarios for AI Doom","utilistrutil","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sNqzGZjv4pRJjjhZs/wild-animal-welfare-scenarios-for-ai-doom",0,"",""],["A note of caution about recent AI risk coverage","Sean_o_h","2023","blog","EA Forum","forum.effectivealtruism.org/posts/weJZjku3HiNgQC4ER/a-note-of-caution-about-recent-ai-risk-coverage",0,"","policy"],["An Exercise to Build Intuitions on AGI Risk","Lauro Langosco","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uKujHaJd2ckAKAevo/an-exercise-to-build-intuitions-on-agi-risk",0,"",""],["Article Summary: Current and Near-Term AI as a Potential Existential Risk Factor","AndreFerretti","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xFXeoYn872J9vr7jh/article-summary-current-and-near-term-ai-as-a-potential-1",0,"",""],["Could AI accelerate economic growth?","Tom_Davidson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aDFR6c3Qd6cqrQu7c/could-ai-accelerate-economic-growth",0,"",""],["Proposal: Tune LLMs to Use Calibrated Language","OneManyNone","2023","blog","LessWrong","www.lesswrong.com/posts/AQDa6HdjjRsGYyJ7Q/proposal-tune-llms-to-use-calibrated-language",0,"",""],["Rethink Priorities is hiring a Compute Governance Researcher or Research Assistant","MichaelA and Rethink Priorities","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PYeMoDripSZsasgi6/rethink-priorities-is-hiring-a-compute-governance-researcher",0,"","governance policy compute-governance"],["The current alignment plan, and how we might improve it | EAG Bay Area 23","Buck","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yCx3kCReJtucpdd33/the-current-alignment-plan-and-how-we-might-improve-it-or",0,"",""],["Understanding how hard alignment is may be the most important research direction right now","Aron","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MwfMqx7EoPjsqtdYK/understanding-how-hard-alignment-is-may-be-the-most",0,"",""],["What will GPT-2030 look like?","jsteinhardt","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WZXqNYbJhtidjRXSi/what-will-gpt-2030-look-like",0,"","forecasting scaling-laws"],["[Linkpost] Given Extinction Worries, Why Don’t AI Researchers Quit? Well, Several Reasons","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Ci2Lh5fuwBqtKSG72/linkpost-given-extinction-worries-why-don-t-ai-researchers",0,"",""],["A Playbook for AI Risk Reduction (focused on misaligned AI)","HoldenKarnofsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fbk9H6ipfybHyqjrp/a-playbook-for-ai-risk-reduction-focused-on-misaligned-ai",0,"",""],["Agentic Mess (A Failure Story)","Karl von Wendt and 5 others","2023","blog","LessWrong","www.lesswrong.com/posts/LyJAFBuuEfd4kxgsw/agentic-mess-a-failure-story",0,"","agents"],["AISN #9: Statement on Extinction Risks, Competitive Pressures, and When Will AI Reach Human-Level?","Center for AI Safety and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BCwaWkMHTMjFMvedS/aisn-9-statement-on-extinction-risks-competitive-pressures",0,"",""],["Algorithmic Improvement Is Probably Faster Than Scaling Now","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CfpAXccrBvWpQw9xj/algorithmic-improvement-is-probably-faster-than-scaling-now",0,"",""],["Rishi to outline his vision for Britain to take the world lead in policing AI threats when he meets Joe Biden","Mati_Roy","2023","blog","LessWrong","www.lesswrong.com/posts/9g5dtLx3KTTvJ6Fb8/rishi-to-outline-his-vision-for-britain-to-take-the-world",0,"","governance"],["Stampy's AI Safety Info - New Distillations #3 [May 2023]","markov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3yAtGF3bCHqkSN52h/stampy-s-ai-safety-info-new-distillations-3-may-2023",0,"",""],["The Sharp Right Turn: sudden deceptive alignment as a convergent goal","avturchin","2023","blog","LessWrong","www.lesswrong.com/posts/Hicfd4C5ffrtEaTbF/the-sharp-right-turn-sudden-deceptive-alignment-as-a",0,"","alignment-faking deception instrumental-convergence"],["Tim Cook was asked about extinction risks from AI","Saul Munn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gZhrqihqSEvbtTBpi/tim-cook-was-asked-about-extinction-risks-from-ai",0,"",""],["Transformative AGI by 2043 is <1% likely","Ted Sanders","2023","blog","LessWrong","www.lesswrong.com/posts/DgzdLzDGsqoRXhCK7/transformative-agi-by-2043-is-less-than-1-likely",0,"","forecasting"],["US Policy Career Resources","US Policy Careers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aSBEN99X2KaRLSmeT/us-policy-career-resources",0,"","governance policy"],["AISafety.info \"How can I help?\" FAQ","steven0461 and Severin T. Seehrich","2023","blog","LessWrong","www.lesswrong.com/posts/nBzTxJmLdebiqhY8q/aisafety-info-how-can-i-help-faq",0,"",""],["Moral Spillover in Human-AI Interaction","Katerina Manoli","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BCoWhBsZbDzaywAdp/moral-spillover-in-human-ai-interaction",0,"",""],["Wildfire of strategicness","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HdA2nDKQ5FJtnTuxP/wildfire-of-strategicness",0,"","agents theory"],["AI Safety Fundamentals: An Informal Cohort Starting Soon!","Tiago de Vassal","2023","blog","LessWrong","www.lesswrong.com/posts/niJmEoLCqgNRe5rtE/ai-safety-fundamentals-an-informal-cohort-starting-soon",0,"",""],["AI Safety Fundamentals: An Informal Cohort Starting Soon! (cross-posted to lesswrong.com)","Tiago","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yJkZK62NKuRu7SaJw/ai-safety-fundamentals-an-informal-cohort-starting-soon",0,"",""],["Decomposing alignment to take advantage of paradigms","Christopher King","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TH2tRumAuwKWN8NoG/decomposing-alignment-to-take-advantage-of-paradigms",0,"",""],["From voluntary to mandatory, are the ESG disclosure frameworks still fertile ground for unrealised EA career pathways? – A 2023 update on ESG potential impact","Christopher Chan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/y4Pu5jhYoRibb9MyC/from-voluntary-to-mandatory-are-the-esg-disclosure",0,"","governance"],["How to Think About Activation Patching","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/xh85KbTFhbCz7taD4/how-to-think-about-activation-patching",0,"","mechanistic-interpretability"],["One implementation of regulatory GPU restrictions","porby","2023","blog","LessWrong","www.lesswrong.com/posts/TjLRJY7gCxDxwM3Cu/one-implementation-of-regulatory-gpu-restrictions",0,"","governance"],["Details on how an IAEA-style AI regulator would function?","freedomandutility","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mYzJxCBCWWZ4rZSsS/details-on-how-an-iaea-style-ai-regulator-would-function",0,"","policy"],["Intrinsic limitations of GPT-4 and other large language models, and why I'm not (very) worried about GPT-n","Fods12","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6dphu3p8d5mQZEZzk/intrinsic-limitations-of-gpt-4-and-other-large-language",0,"","forecasting"],["Terry Tao is hosting an \"AI to Assist Mathematical Reasoning\" workshop","junk heap homotopy","2023","blog","LessWrong","www.lesswrong.com/posts/a5EK9WTv6x8htkGXW/terry-tao-is-hosting-an-ai-to-assist-mathematical-reasoning",0,"",""],["The AGI Race Between the US and China Doesn’t Exist.","Eva_B","2023","blog","LessWrong","www.lesswrong.com/posts/z4MDDwwnWKnv2ZzdK/the-agi-race-between-the-us-and-china-doesn-t-exist",0,"","governance"],["Unfaithful Explanations in Chain-of-Thought Prompting","miles","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6eKL9wDqeiELbKPDj/unfaithful-explanations-in-chain-of-thought-prompting",0,"","chain-of-thought-faithfulness"],["Upcoming AI regulations are likely to make for an unsafer world","shminux","2023","blog","LessWrong","www.lesswrong.com/posts/g2aeGupbr3XC68tLJ/upcoming-ai-regulations-are-likely-to-make-for-an-unsafer",0,"","governance"],["[Replication] Conjecture's Sparse Coding in Toy Models","Hoagy and Logan Riggs","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/E8imxQo96WgDCMxkA/replication-conjecture-s-sparse-coding-in-toy-models",0,"",""],["Advice for Entering AI Safety Research","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/HCZ6feW2EGXuiwuid/advice-for-entering-ai-safety-research",0,"",""],["Applications open for AI Safety Fundamentals: Governance Course","Jamie Bernardi and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bsbf4am9paoTq8Lrb/applications-open-for-ai-safety-fundamentals-governance",0,"","governance"],["Catastrophic Risks from Unsafe AI: Navigating a Tightrope Scenario (Ben Garfinkel, EAG London 2023)","AlexanderSaeri","2023","blog","EA Forum","forum.effectivealtruism.org/posts/goYTp3CyLA4dnL2kN/catastrophic-risks-from-unsafe-ai-navigating-a-tightrope",0,"","governance policy"],["Co-found an incubator for independent AI Safety researchers (rolling applications)","Alexandra Bos","2023","blog","LessWrong","www.lesswrong.com/posts/jA3HfgcdiT2DLhvxp/co-found-an-incubator-for-independent-ai-safety-researchers",0,"",""],["Inference from a Mathematical Description of an Existing Alignment Research: a proposal for an outer alignment research program","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/rnzpYKWmNWEW5PQyq/inference-from-a-mathematical-description-of-an-existing",0,"",""],["Proposal: labs should precommit to pausing if an AI argues for itself to be improved","NickGabs","2023","blog","LessWrong","www.lesswrong.com/posts/nbrPmaDwvKQD8gdFj/proposal-labs-should-precommit-to-pausing-if-an-ai-argues",0,"","alignment-faking deception governance forecasting"],["Some thoughts on \"AI could defeat all of us combined\"","Milan_Griffes","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9oDMuY2cGfqBfp94T/some-thoughts-on-ai-could-defeat-all-of-us-combined",0,"",""],["The Control Problem: Unsolved or Unsolvable?","Remmelt","2023","blog","LessWrong","www.lesswrong.com/posts/xp6n2MG5vQkPpFEBH/the-control-problem-unsolved-or-unsolvable",0,"",""],["Think carefully before calling RL policies \"agents\"","TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rmfjo4Wmtgq8qa2B7/think-carefully-before-calling-rl-policies-agents",0,"","agents"],["AI Manufactured Crisis (don't trust AI to protect us from AI)","WobblyPanda2","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sCTJdSavgXkDo8Log/ai-manufactured-crisis-don-t-trust-ai-to-protect-us-from-ai",0,"",""],["An explanation of decision theories","metachirality","2023","blog","LessWrong","www.lesswrong.com/posts/RiYhceiQy4w8JQAsn/an-explanation-of-decision-theories",0,"","theory"],["Four levels of understanding decision theory","Max H","2023","blog","LessWrong","www.lesswrong.com/posts/sx47Wi2x8c4mqkYAC/four-levels-of-understanding-decision-theory",0,"","theory"],["How will they feed us","meijer1973","2023","blog","LessWrong","www.lesswrong.com/posts/B6LvjefPmHdBFts4z/how-will-they-feed-us",0,"",""],["Open Source LLMs Can Now Actively Lie","Josh Levy","2023","blog","LessWrong","www.lesswrong.com/posts/fknxaKAqrc6hguvJD/open-source-llms-can-now-actively-lie",0,"","alignment-faking deception"],["Outreach success: Intro to AI risk that has been successful","Michael Tontchev","2023","blog","LessWrong","www.lesswrong.com/posts/B8Djo44WtZK6kK4K5/outreach-success-intro-to-ai-risk-that-has-been-successful",0,"",""],["Primitive Global Discourse Framework, Constitutional AI using legal frameworks, and Monoculture - A loss of control over the role of AGI in society","broptross","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZMd2hjMF2auyeqtfr/primitive-global-discourse-framework-constitutional-ai-using",0,"","constitutional-ai"],["Safe AI and moral AI","William D'Alessandro","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FNGcnxAjezjZQqSzv/safe-ai-and-moral-ai",0,"",""],["Short Remark on the (subjective) mathematical 'naturalness' of the Nanda--Lieberum addition modulo 113 algorithm","Spencer Becker-Kahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tdENX8dzdro8PXAzP/short-remark-on-the-subjective-mathematical-naturalness-of",0,"","interpretability"],["Uncertainty about the future does not imply that AGI will go well","Lauro Langosco","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/x5aTiznxJ4o9EGdj9/uncertainty-about-the-future-does-not-imply-that-agi-will-go",0,"","forecasting"],["Update from Campaign for AI Safety","Nik Samoylov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/N6dyo3cyk7eCLAoEB/update-from-campaign-for-ai-safety",0,"",""],["Yes, avoiding extinction from AI *is* an urgent priority: a response to Seth Lazar, Jeremy Howard, and Arvind Narayanan.","Soroush Pour","2023","blog","LessWrong","www.lesswrong.com/posts/MSGMeKgPLrnMyPJYy/yes-avoiding-extinction-from-ai-is-an-urgent-priority-a",0,"",""],["A compute-based framework for thinking about the future of AI","Matthew_Barnett","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fsaogRokXxby6LFd7/a-compute-based-framework-for-thinking-about-the-future-of",0,"","governance compute-governance forecasting"],["A moral backlash against AI will probably slow down AGI development","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/veR4W92bZsTsGgS3D/a-moral-backlash-against-ai-will-probably-slow-down-agi",0,"","policy forecasting"],["A push towards interactive transformer decoding","R0bk","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5msxxQiTDmcDNBnkF/a-push-towards-interactive-transformer-decoding",0,"",""],["Considerations on transformative AI and explosive growth from a semiconductor-industry perspective","Muireall","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XTNwCtsecACuARTcH/considerations-on-transformative-ai-and-explosive-growth",0,"","forecasting"],["Contrast Pairs Drive the Empirical Performance of Contrast Consistent Search (CCS)","Scott Emmons","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9vwekjD6xyuePX7Zr/contrast-pairs-drive-the-empirical-performance-of-contrast",0,"","interpretability eliciting-latent-knowledge"],["Cosmopolitan values don't come free","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2NncxDQ3KBDCxiJiP/cosmopolitan-values-don-t-come-free",0,"",""],["Exponential AI takeoff is a myth","Christoph Hartmann","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wY6aBzcXtSprmDhFN/exponential-ai-takeoff-is-a-myth",0,"","forecasting"],["Improving mathematical reasoning with process supervision","Bowen Baker and 6 others","2023","blog","openai.com","openai.com/research/improving-mathematical-reasoning-with-process-supervision",0,"",""],["Intent-aligned AI systems deplete human agency: the need for agency foundations research in AI safety","catubc","2023","blog","LessWrong","www.lesswrong.com/posts/dDDi9bZm6ELSXTJd9/intent-aligned-ai-systems-deplete-human-agency-the-need-for",0,"","agents theory"],["Limiting factors to predict AI take-off speed","Alfonso Pérez Escudero","2023","blog","LessWrong","www.lesswrong.com/posts/p5ifq7Njn86mdDiHH/limiting-factors-to-predict-ai-take-off-speed",0,"","forecasting"],["My AI-risk cartoon","pre","2023","blog","LessWrong","www.lesswrong.com/posts/RBghcnNGqQFy49YMh/my-ai-risk-cartoon",0,"",""],["Neuroevolution, Social Intelligence, and Logic","vinnik.dmitry07","2023","blog","LessWrong","www.lesswrong.com/posts/rsar32pysCXCTikdd/neuroevolution-social-intelligence-and-logic",0,"","theory"],["Shutdown-Seeking AI","Simon Goldstein","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FgsoWSACQfyyaB5s7/shutdown-seeking-ai",0,"",""],["The EU AI Act needs a definition of high-risk foundation models to avoid regulatory overreach and backlash","matthias_samwald","2023","blog","EA Forum","forum.effectivealtruism.org/posts/p7qXjisiADiCBnofk/the-eu-ai-act-needs-a-definition-of-high-risk-foundation",0,"","evals governance policy"],["Unpredictability and the Increasing Difficulty of AI Alignment for Increasingly Intelligent AI","Max_He-Ho","2023","blog","LessWrong","www.lesswrong.com/posts/BmhtnubpEXsTyi37h/unpredictability-and-the-increasing-difficulty-of-ai",0,"",""],["Advice for new alignment people: Info Max","Jonas Hallgren","2023","blog","LessWrong","www.lesswrong.com/posts/2yxg5RNJ77yCFffMg/advice-for-new-alignment-people-info-max",0,"",""],["AI Doom and David Hume: A Defence of Empiricism in AI Safety","Matt Beard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MDkYSuCzFbEgGgtAd/ai-doom-and-david-hume-a-defence-of-empiricism-in-ai-safety",0,"",""],["AI Safety Newsletter #8: Rogue AIs, how to screen for AI risks, and grants for research on democratic governance of AI","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Fw7wtyCZAaJdioKWE/ai-safety-newsletter-8-rogue-ais-how-to-screen-for-ai-risks",0,"","evals governance"],["Announcing Apollo Research","Marius Hobbhahn and 6 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FG6icLPKizEaWHex5/announcing-apollo-research",0,"","interpretability evals alignment-faking deception governance"],["Boomerang - protocol to dissolve some commitment races","Filip Sondej","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/H9KekSfzHnPLTz4DE/boomerang-protocol-to-dissolve-some-commitment-races-2",0,"",""],["Implications of AGI on Subjective Human Experience","Erica S.","2023","blog","EA Forum","forum.effectivealtruism.org/posts/F8B4JTgfDMXDd7q7G/implications-of-agi-on-subjective-human-experience",0,"",""],["LIMA: Less Is More for Alignment","Ulisse Mini","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dJumQtpoKhjDKH9q8/lima-less-is-more-for-alignment",0,"",""],["PaLM-2 & GPT-4 in \"Extrapolating GPT-N performance\"","Lukas Finnveden","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/75o8oja43LXGAqbAR/palm-2-and-gpt-4-in-extrapolating-gpt-n-performance",0,"","forecasting"],["Statement on AI Extinction - Signed by AGI Labs, Top Academics, and Many Other Notable Figures","Dan H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HcJPJxkyCsrpSdCii/statement-on-ai-extinction-signed-by-agi-labs-top-academics",0,"","governance"],["Statement on AI Extinction - Signed by AGI Labs, Top Academics, and Many Other Notable Figures","Center for AI Safety","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Yk4D4DZpx6eriMDyY/statement-on-ai-extinction-signed-by-agi-labs-top-academics",0,"",""],["Statement on AI Risk","Center for AI Safety","2023","report","safe.ai","www.safe.ai/statement-on-ai-risk",0,"",""],["The bullseye framework: My case against AI doom","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Nxtq2d8Xb3QuuHKE8/the-bullseye-framework-my-case-against-ai-doom",0,"",""],["The Case for AI Adaptation: The Perils of Living in a World with Aligned and Well-Deployed Transformative Artificial Intelligence","HTC","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bFDs7yFiEhgPt4LWt/the-case-for-ai-adaptation-the-perils-of-living-in-a-world",0,"","governance"],["The case for removing alignment and ML research from the training dataset","beren","2023","blog","LessWrong","www.lesswrong.com/posts/XnnMYMjDGuYhkhQPB/the-case-for-removing-alignment-and-ml-research-from-the",0,"","governance training-data"],["Who is liable for AI?","jmh","2023","blog","LessWrong","www.lesswrong.com/posts/hAJgbMZydoQJxLnMD/who-is-liable-for-ai",0,"","governance"],["Aligning an H-JEPA agent via training on the outputs of an LLM-based \"exemplary actor\"","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MJXwnHbqFYE3N4dP2/aligning-an-h-jepa-agent-via-training-on-the-outputs-of-an",0,"","interpretability agents"],["An LLM-based “exemplary actor”","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4ztqncYBakD6DWuXC/an-llm-based-exemplary-actor",0,"","automated-alignment-research"],["Import AI 331: 16X smaller language models; could AMD compete with NVIDIA?; and BERT for the dark web","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-331-16x-smaller-language",0,"",""],["Language Agents Reduce the Risk of Existential Catastrophe","cdkg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6odj5iN8zoDL3t224/language-agents-reduce-the-risk-of-existential-catastrophe",0,"","agents"],["List of Masters Programs in Tech Policy, Public Policy and Security (Europe)","sberg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8CD4i8FsRApcbt3an/list-of-masters-programs-in-tech-policy-public-policy-and",0,"","governance policy"],["Sentience matters","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Htu55gzoiYHS6TREB/sentience-matters",0,"",""],["What are some of the best introductions/breakdowns of AI existential risk for those unfamiliar?","Isaac King","2023","blog","LessWrong","www.lesswrong.com/posts/ZcAocGh9Ck4o2HCeh/what-are-some-of-the-best-introductions-breakdowns-of-ai",0,"",""],["Wikipedia as an introduction to the alignment problem","SoerenMind","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bAmLMbyzXK3HtztbB/wikipedia-as-an-introduction-to-the-alignment-problem",0,"",""],["Without a trajectory change, the development of AGI is likely to go badly","Max H","2023","blog","LessWrong","www.lesswrong.com/posts/qrrEtrbLcmqr3b5uf/without-a-trajectory-change-the-development-of-agi-is-likely",0,"","forecasting"],["Language Agents Reduce the Risk of Existential Catastrophe","cdkg and Simon Goldstein","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8hf5hNksjn78CouKR/language-agents-reduce-the-risk-of-existential-catastrophe",0,"","agents chain-of-thought-faithfulness"],["My AI Alignment Research Agenda and Threat Model, right now (May 2023)","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/PA2hprrtvtpPMugeN/my-ai-alignment-research-agenda-and-threat-model-right-now",0,"",""],["Status Quo Engines - AI essay","Ilana_Goldowitz_Jimenez","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Xdf3cnuSR68btiTKR/status-quo-engines-ai-essay",0,"",""],["TinyStories: Small Language Models That Still Speak Coherent English","Ulisse Mini","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dMoaBvcxpBE7LcES4/tinystories-small-language-models-that-still-speak-coherent",0,"",""],["Why and When Interpretability Work is Dangerous","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/uhMRgEXabYbWeLc6T/why-and-when-interpretability-work-is-dangerous",0,"","interpretability"],["[Linkpost] Longtermists Are Pushing a New Cold War With China","Mohammad Ismam Huda","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5wwcMr8tDqCwZrDGM/linkpost-longtermists-are-pushing-a-new-cold-war-with-china",0,"",""],["By failing to take serious AI action, the US could be in violation of its international law obligations","Cecil Abungu","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7yjd2wJjSqbzz3dZX/by-failing-to-take-serious-ai-action-the-us-could-be-in",0,"","governance policy"],["Diminishing Returns in Machine Learning Part 1: Hardware Development and the Physical Frontier","Brian Chau","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wdxMmSnK5JscvuK35/diminishing-returns-in-machine-learning-part-1-hardware",0,"","forecasting"],["Hands-On Experience Is Not Magic","Thane Ruthenis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tNtiJp8dA6jMbgKbf/hands-on-experience-is-not-magic",0,"",""],["Is Deontological AI Safe? [Feedback Draft]","Dan H and William D'Alessandro","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gbNqWpDwmrWmzopQW/is-deontological-ai-safe-feedback-draft",0,"",""],["Project Idea: Challenge Groups for Alignment Researchers","Adam Zerner","2023","blog","LessWrong","www.lesswrong.com/posts/Xni8DSjkK5BxSJbiF/project-idea-challenge-groups-for-alignment-researchers",0,"",""],["[Job Ad] SERI MATS is hiring for our summer program","zanekay","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YsnNxGxFaQnvg63AQ/job-ad-seri-mats-is-hiring-for-our-summer-program",0,"",""],["[Linkpost] OpenAI is awarding ten 100k grants for building prototypes of a democratic process for steering AI","pseudonym","2023","blog","EA Forum","forum.effectivealtruism.org/posts/b8xQEHeyABqv9ft53/linkpost-openai-is-awarding-ten-100k-grants-for-building",0,"","governance policy"],["Bandgaps, Brains, and Bioweapons: The limitations of computational science and what it means for AGI","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vbGKuNsS5ix5g7Nqk/bandgaps-brains-and-bioweapons-the-limitations-of",0,"",""],["Before smart AI, there will be many mediocre or specialized AIs","Lukas Finnveden","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5sWNnbHRkExfLaS49/before-smart-ai-there-will-be-many-mediocre-or-specialized",0,"","forecasting"],["Biomimetic alignment: Alignment between animal genes and animal brains as a model for alignment between humans and AI systems.","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tQxLfkvFGWBhJ2KzR/biomimetic-alignment-alignment-between-animal-genes-and",0,"",""],["Conditional Prediction with Zero-Sum Training Solves Self-Fulfilling Prophecies","Rubi J. Hudson and Johannes Treutlein","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/A48amesEmqD8KNSmY/conditional-prediction-with-zero-sum-training-solves-self",0,"",""],["how humans are aligned","bhauth","2023","blog","LessWrong","www.lesswrong.com/posts/NptxTqHDtFovhtW9b/how-humans-are-aligned-1",0,"",""],["Some thoughts on automating alignment research","Lukas Finnveden","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KwQYsF4XFtPqjgwvH/some-thoughts-on-automating-alignment-research-1",0,"",""],["What's your viewpoint on the likelihood of GPT-5 being able to autonomously create, train, and implement an AI superior to GPT-5?","Super AGI","2023","blog","LessWrong","www.lesswrong.com/posts/8cwwtEzeiFGZpBiwt/what-s-your-viewpoint-on-the-likelihood-of-gpt-5-being-able",0,"","forecasting robustness"],["[Linkpost] OpenAI leaders call for regulation of \"superintelligence\" to reduce existential risk.","Lowe","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2sepfMDwgRfBpQC8S/linkpost-openai-leaders-call-for-regulation-of",0,"","governance policy"],["An early warning system for novel AI risks","Toby Shevlane","2023","blog","deepmind.com","www.deepmind.com/blog/an-early-warning-system-for-novel-ai-risks",0,"",""],["DeepMind: Model evaluation for extreme risks","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/FdQzArWhERh4YZqY9/deepmind-model-evaluation-for-extreme-risks",0,"","evals"],["Exploiting Newcomb's Game Show","carterallen","2023","blog","LessWrong","www.lesswrong.com/posts/AhANWF2Y4SXYeN5Wr/exploiting-newcomb-s-game-show",0,"","alignment-faking deception theory"],["Is behavioral safety \"solved\" in non-adversarial conditions?","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/ZYddmLsTGaTLdXGwj/is-behavioral-safety-solved-in-non-adversarial-conditions",0,"","rlhf"],["Requirements for a STEM-capable AGI Value Learner (my Case for Less Doom)","RogerDearnaley","2023","blog","LessWrong","www.lesswrong.com/posts/q9yPYG2St2L4SEtKW/requirements-for-a-stem-capable-agi-value-learner-my-case-1",0,"","goodharts-law instrumental-convergence automated-alignment-research forecasting chain-of-thought-faithfulness theory"],["Solving the Mechanistic Interpretability challenges: EIS VII Challenge 2","StefanHex and Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/k43v47eQjaj6fY7LE/solving-the-mechanistic-interpretability-challenges-eis-vii-1",0,"","interpretability mechanistic-interpretability"],["The Genie in the Bottle: An Introduction to AI Alignment and Risk","Snorkelfarsan","2023","blog","LessWrong","www.lesswrong.com/posts/menRJyuyc5yzGdTGf/the-genie-in-the-bottle-an-introduction-to-ai-alignment-and",0,"",""],["Two ideas for alignment, perpetual mutual distrust and induction","APaleBlueDot","2023","blog","LessWrong","www.lesswrong.com/posts/jpLJdFMGJiKKBNoLy/two-ideas-for-alignment-perpetual-mutual-distrust-and",0,"","governance"],["Will AI end everything? A guide to guessing | EAG Bay Area 23","Katja_Grace","2023","blog","EA Forum","forum.effectivealtruism.org/posts/whEmrvK9pzioeircr/will-ai-end-everything-a-guide-to-guessing-or-eag-bay-area",0,"","forecasting"],["[Linkpost] Interpretability Dreams","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DQ4y5tvotag5KPzcu/linkpost-interpretability-dreams",0,"","interpretability"],["AGI Catastrophe and Takeover: Some Reference Class-Based Priors","zdgroff","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MDNcMLQfxg2n9qXEZ/agi-catastrophe-and-takeover-some-reference-class-based",0,"","forecasting"],["Aligned AI via monitoring objectives in AutoGPT-like systems","Paul Colognese","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pihmQv5XezwkxJk2a/aligned-ai-via-monitoring-objectives-in-autogpt-like-systems",0,"","monitoring chain-of-thought-faithfulness"],["Diagram with Commentary for AGI as an X-Risk","Jared Leibowich","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PS5GKpvKxDhPCcg7y/diagram-with-commentary-for-agi-as-an-x-risk",0,"","forecasting"],["New s-risks audiobook available now","Alistair Webster","2023","blog","EA Forum","forum.effectivealtruism.org/posts/s5W3FTYYoR4hvnL8h/new-s-risks-audiobook-available-now",0,"",""],["October 2022 AI Risk Community Survey Results","Froolow","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tyneYFeDqBgXYykxG/october-2022-ai-risk-community-survey-results",0,"",""],["Rishi Sunak mentions \"existential threats\" in talk with OpenAI, DeepMind, Anthropic CEOs","Arjun Panickssery and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/2PpRAXRbNf3rrbF9X/rishi-sunak-mentions-existential-threats-in-talk-with-openai",0,"","governance"],["What projects and efforts are there to promote AI safety research?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/gvfRr2TsKao4trDWw/what-projects-and-efforts-are-there-to-promote-ai-safety",0,"",""],["'Fundamental' vs 'applied' mechanistic interpretability research","Lee Sharkey","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uvEyizLAGykH8LwMx/fundamental-vs-applied-mechanistic-interpretability-research",0,"","interpretability mechanistic-interpretability"],["[Linkpost] The AGI Show podcast","Soroush Pour","2023","blog","LessWrong","www.lesswrong.com/posts/KdEkNx3SgjfciNKPx/linkpost-the-agi-show-podcast",0,"",""],["A Different Approach to Community Building: The Spiral Path to Impact","ezrah","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9apBqe4KH394cS89z/a-different-approach-to-community-building-the-spiral-path",0,"",""],["AI Safety Newsletter #7: Disinformation, Governance Recommendations for AI labs, and Senate Hearings on AI","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/B3akyeohHhinHkGZM/ai-safety-newsletter-7-disinformation-governance",0,"","governance policy"],["AI Safety Newsletter #7: Disinformation, Governance Recommendations for AI labs, and Senate Hearings on AI","Dan H and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/rniJ8tPweahjDKJWM/ai-safety-newsletter-7-disinformation-governance",0,"","governance"],["AI self-improvement is possible","bhauth","2023","blog","LessWrong","www.lesswrong.com/posts/rSycgquipFkozDHzF/ai-self-improvement-is-possible",0,"","forecasting"],["Data and \"tokens\" a 30 year old human \"trains\" on","Jose Miguel Cruz y Celis","2023","blog","LessWrong","www.lesswrong.com/posts/ftEvHLAXia8Cm9W5a/data-and-tokens-a-30-year-old-human-trains-on",0,"","scaling-laws"],["How I learned to stop worrying and love skill trees","Clark Urzo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6EQd9F2md4d7dGpT2/how-i-learned-to-stop-worrying-and-love-skill-trees",0,"",""],["How I learned to stop worrying and love skill trees","junk heap homotopy","2023","blog","LessWrong","www.lesswrong.com/posts/3CsynkTxNEdHDexTT/how-i-learned-to-stop-worrying-and-love-skill-trees",0,"",""],["Is \"brittle alignment\" good enough?","the8thbit","2023","blog","LessWrong","www.lesswrong.com/posts/z3NFWLwXYdJui7mg6/is-brittle-alignment-good-enough",0,"","robustness"],["Some governance research ideas to prevent malevolent control over AGI and why this might matter a hell of a lot","Jim Buhler","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TLSPQjjXZruwmg4PE/some-governance-research-ideas-to-prevent-malevolent-control",0,"","governance"],["The Polarity Problem [Draft]","Dan H and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/idcnnZGEPfxuaSPBx/the-polarity-problem-draft",0,"","forecasting"],["Will Artificial Superintelligence Kill Us?","James_Miller","2023","blog","LessWrong","www.lesswrong.com/posts/mkaaLsuCGJwiYzpig/will-artificial-superintelligence-kill-us",0,"",""],["🐶Safetensors audited as really safe and becoming the default","Nicolas Patry and 2 others","2023","blog","blog.eleuther.ai","blog.eleuther.ai/safetensors-security-audit/",0,"",""],["[Linkpost] \"Governance of superintelligence\" by OpenAI","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oo96uRHNbGjr4DHut/linkpost-governance-of-superintelligence-by-openai",0,"","governance"],["Activation additions in a small residual network","Garrett Baker","2023","blog","LessWrong","www.lesswrong.com/posts/mAMxGxSC94BqCi9aJ/activation-additions-in-a-small-residual-network",0,"","interpretability"],["AI Safety in China: Part 2","Lao Mein","2023","blog","LessWrong","www.lesswrong.com/posts/zjym2uaPsg9n3EjY6/ai-safety-in-china-part-2",0,"",""],["AI strategy career pipeline","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gSGhrCXdntxLrMAmJ/ai-strategy-career-pipeline",0,"","governance"],["Conjecture internal survey: AGI timelines and probability of human extinction from advanced AI","Maris Sala","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kygEPBDrGGoM8rz9a/conjecture-internal-survey-agi-timelines-and-probability-of",0,"","forecasting"],["Distillation of Neurotech and Alignment Workshop January 2023","lisathiergart and Sumner L Norman","2023","blog","LessWrong","www.lesswrong.com/posts/KQSpRoQBz7f6FcXt3/distillation-of-neurotech-and-alignment-workshop-january-1",0,"",""],["How Rogue AIs may Arise","Yoshua Bengio","2023","report","yoshuabengio.org","yoshuabengio.org/2023/05/22/how-rogue-ais-may-arise/",0,"",""],["I don't want to talk about ai","Kirsten and EA Lifestyles","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XYC8jmM4WPCDYZZmm/i-don-t-want-to-talk-about-ai",0,"",""],["Import AI 330: Palantir's AI-War future; BLOOMChat; and more money for distributed AI training","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-330-palantirs-ai-war-future",0,"",""],["When will digital compute match the human brain?","Yarrow Bouchard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bzchdNZma9TaXY46Y/when-will-digital-compute-match-the-human-brain",0,"","forecasting"],["Former Israeli Prime Minister Speaks About AI X-Risk","Yonatan Cale","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Konde3tJY2SFoFbpd/former-israeli-prime-minister-speaks-about-ai-x-risk",0,"",""],["Effective Altruism Florida's AI Expert Panel - Recording and Slides Available","Sam Emerman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tRjZsXK2zWifdfCKN/effective-altruism-florida-s-ai-expert-panel-recording-and",0,"","policy"],["G7 Summit—Cooperation on AI Policy","Lenny","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6diKmxL8hD89yoMrq/g7-summit-cooperation-on-ai-policy",0,"","policy"],["Mr. Meeseeks as an AI capability tripwire","Eric Zhang","2023","blog","LessWrong","www.lesswrong.com/posts/7yEFHisCQSCpLnqWQ/mr-meeseeks-as-an-ai-capability-tripwire",0,"",""],["The Compleat Cybornaut","ukc10014 and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/iFBdEqEogtXcjCPBB/the-compleat-cybornaut",0,"","rlhf evals"],["“The Race to the End of Humanity” – Structural Uncertainty Analysis in AI Risk Models","Froolow","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JjAjJ53mmpQqBeobQ/the-race-to-the-end-of-humanity-structural-uncertainty",0,"",""],["A recent write-up of the case for AI (existential) risk","Timsey","2023","blog","EA Forum","forum.effectivealtruism.org/posts/khHGQBH7yk2rGWqLX/a-recent-write-up-of-the-case-for-ai-existential-risk",0,"",""],["AI #12:The Quest for Sane Regulations","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/5nDxmAvZ9w5CPa9gR/ai-12-the-quest-for-sane-regulations",0,"","governance"],["Asking for online resources why AI now is near AGI","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LsZbHLYEpogShDb6a/asking-for-online-resources-why-ai-now-is-near-agi",0,"","forecasting"],["Collective Identity","NicholasKees and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gLyRQCg6kp5cqTQTm/collective-identity",0,"",""],["Some background for reasoning about dual-use alignment research","Charlie Steiner","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zswuToWK6zpYSwmCn/some-background-for-reasoning-about-dual-use-alignment",0,"",""],["The Unexpected Clanging","Chris_Leong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hn65bRY3BoXEW3bxL/the-unexpected-clanging",0,"","theory"],["Thread: Reflections on the AGI Safety Fundamentals course?","Clifford","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TbfioWKYYWPT2Dgue/thread-reflections-on-the-agi-safety-fundamentals-course",0,"",""],["We Shouldn't Expect AI to Ever be Fully Rational","OneManyNone","2023","blog","LessWrong","www.lesswrong.com/posts/KYXHneyrnNNHLKWGJ/we-shouldn-t-expect-ai-to-ever-be-fully-rational",0,"",""],["AI Alignment in The New Yorker","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/54RWjNAn2hqeo3Xzq/ai-alignment-in-the-new-yorker-1",0,"",""],["Creating a self-referential system prompt for GPT-4","Ozyrus","2023","blog","LessWrong","www.lesswrong.com/posts/vn9huEHsCGEQzTfrW/creating-a-self-referential-system-prompt-for-gpt-4",0,"","chain-of-thought-faithfulness"],["Eisenhower's Atoms for Peace Speech","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/bdsCWKKDSnh8wNKS3/eisenhower-s-atoms-for-peace-speech",0,"","governance"],["GPT-4 implicitly values identity preservation: a study of LMCA identity management","Ozyrus","2023","blog","LessWrong","www.lesswrong.com/posts/tJzAHPFWFnpbL5a3H/gpt-4-implicitly-values-identity-preservation-a-study-of",0,"","chain-of-thought-faithfulness"],["Lessons on project management from “How Big Things Get Done”","Cristina Schmidt Ibáñez","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oHZPackrDrjyEb9Sk/lessons-on-project-management-from-how-big-things-get-done",0,"","forecasting"],["Let’s use AI to harden human defenses against AI manipulation","Tom Davidson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zxmzBTwKkPMxQQcfR/let-s-use-ai-to-harden-human-defenses-against-ai",0,"",""],["Play Regrantor: Move up to $250,000 to Your Top High-Impact Projects!","Dawn Drescher and Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Lna7SayJkyrKczH4n/play-regrantor-move-up-to-usd250-000-to-your-top-high-impact",0,"","forecasting"],["Some quotes from Tuesday's Senate hearing on AI","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kXaxasXfG8DQR4jgq/some-quotes-from-tuesday-s-senate-hearing-on-ai",0,"","policy"],["Why AGI systems will not be fanatical maximisers (unless trained by fanatical humans)","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/j9yT9Sizu2sjNuygR/why-agi-systems-will-not-be-fanatical-maximisers-unless",0,"",""],["$500 Bounty/Prize Problem: Channel Capacity Using \"Insensitive\" Functions","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YDSjSpD7yBoivMHay/usd500-bounty-prize-problem-channel-capacity-using",0,"",""],["A Mechanistic Interpretability Analysis of a GridWorld Agent-Simulator (Part 1 of N)","Joseph Bloom","2023","blog","LessWrong","www.lesswrong.com/posts/JvQWbrbPjuvw4eqxv/a-mechanistic-interpretability-analysis-of-a-gridworld-agent",0,"","interpretability mechanistic-interpretability agents"],["AI Risk & Policy Forecasts from Metaculus & FLI's AI Pathways Workshop","Will Aldred","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Ewk9eXrcRRcJvqBY8/ai-risk-and-policy-forecasts-from-metaculus-and-fli-s-ai",0,"","governance policy forecasting"],["AI Risk & Policy Forecasts from Metaculus & FLI's AI Pathways Workshop","_will_","2023","blog","LessWrong","www.lesswrong.com/posts/eK9SwXrSY4s7p2RgY/ai-risk-and-policy-forecasts-from-metaculus-and-fli-s-ai",0,"","governance policy forecasting"],["AI Safety Newsletter #6: Examples of AI safety progress, Yoshua Bengio proposes a ban on AI agents, and lessons from nuclear arms control","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SxpQpqXBvAxrPWC2e/ai-safety-newsletter-6-examples-of-ai-safety-progress-yoshua",0,"","agents"],["AI Safety Newsletter #6: Examples of AI safety progress, Yoshua Bengio proposes a ban on AI agents, and lessons from nuclear arms control","Dan H and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/e2XAqFyEBWxzGXeHy/ai-safety-newsletter-6-examples-of-ai-safety-progress-yoshua",0,"","agents"],["AI Will Not Want to Self-Improve","petersalib","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/axKWaxjc2CHH5gGyN/ai-will-not-want-to-self-improve",0,"",""],["Decision Theory with the Magic Parts Highlighted","moridinamael","2023","blog","LessWrong","www.lesswrong.com/posts/5sFkZK342j5CmBCm8/decision-theory-with-the-magic-parts-highlighted",0,"","theory"],["Evaluating Language Model Behaviours for Shutdown Avoidance in Textual Scenarios","Simon Lermen and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/BQm5wgtJirrontgRt/evaluating-language-model-behaviours-for-shutdown-avoidance",0,"","evals"],["My current workflow to study the internal mechanisms of LLM","Yulu Pi","2023","blog","LessWrong","www.lesswrong.com/posts/eXPSTu9u8uCvERKnq/my-current-workflow-to-study-the-internal-mechanisms-of-llm",0,"","interpretability"],["OpenAI CEO Sam Altman testifies during Senate hearing on AI oversight — 05/16/23","CNBC Television","2023","report","youtube.com","www.youtube.com/watch?v=fP5YdyjTfG0",0,"",""],["Oversight of A.I.: Rules for Artificial Intelligence","US Senate Committee on the Judiciary","2023","report","judiciary.senate.gov","www.judiciary.senate.gov/committee-activity/hearings/oversight-of-ai-rules-for-artificial-intelligence",0,"",""],["Proposal: we should start referring to the risk from unaligned AI as a type of *accident risk*","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/marPAMc9yCWG79h7f/proposal-we-should-start-referring-to-the-risk-from",0,"",""],["Why doesn't the presence of log-loss for probabilistic models (e.g. sequence prediction) imply that any utility function capable of producing a \"fairly capable\" agent will have at least some non-negligible fraction of overlap with human values?","Thoth Hermes","2023","blog","LessWrong","www.lesswrong.com/posts/JPr9qcBR4SpC93YeG/why-doesn-t-the-presence-of-log-loss-for-probabilistic",0,"","agents"],["Accidentally teaching AI models to deceive us (Ajeya Cotra on The 80,000 Hours Podcast)","80000_Hours and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tX3ax2aSTbu4BtQBN/accidentally-teaching-ai-models-to-deceive-us-ajeya-cotra-on",0,"",""],["AI policy & governance in Australia: notes from an initial discussion","AlexanderSaeri","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pq5tS6WGmeaTWi5uu/ai-policy-and-governance-in-australia-notes-from-an-initial",0,"","governance policy"],["Can we learn much by studying the behaviour of RL policies?","AidanGoth","2023","blog","LessWrong","www.lesswrong.com/posts/RtbcxXkFwAaDBtQGP/can-we-learn-much-by-studying-the-behaviour-of-rl-policies",0,"","theory"],["Catastrophic Regressional Goodhart: Appendix","Thomas Kwa and Drake Thomas","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GdkixRevWpEanYgou/catastrophic-regressional-goodhart-appendix",0,"","goodharts-law"],["EA and AI Safety Schism: AGI, the last tech humans will (soon*) build","Phib","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ayWPwLRjxecTLEDkN/ea-and-ai-safety-schism-agi-the-last-tech-humans-will-soon",0,"","forecasting"],["GovAI: Towards best practices in AGI safety and governance: A survey of expert opinion","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/fkqvztgszJpqmDHom/govai-towards-best-practices-in-agi-safety-and-governance-a",0,"","governance"],["Import AI 329: Compute IS data; don't build AI agents; AI needs a precautionary principle","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-329-compute-is-data-dont",0,"","agents"],["Reward is the optimization target (of capabilities researchers)","Max H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/w6c47JGY3C4k4dWBc/reward-is-the-optimization-target-of-capabilities-1",0,"",""],["Simple experiments with deceptive alignment","Andreas_Moe","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8dJ8LgpjzWJQfHAfx/simple-experiments-with-deceptive-alignment-1",0,"","alignment-faking deception"],["Some Summaries of Agent Foundations Work","mattmacdermott","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3vDb6EzBpaHqDqQif/some-summaries-of-agent-foundations-work-1",0,"","agents theory"],["The Lightcone Theorem: A Better Foundation For Natural Abstraction?","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2WuSZo7esdobiW2mr/the-lightcone-theorem-a-better-foundation-for-natural",0,"",""],["Un-unpluggability - can't we just unplug it?","Oliver Sourbut","2023","blog","LessWrong","www.lesswrong.com/posts/cniLbC8EFf777Aspb/un-unpluggability-can-t-we-just-unplug-it",0,"","governance forecasting"],["Why don't quantilizers also cut off the upper end of the distribution?","Alex_Altair","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RHdoMEbP8MxeAmQo5/why-don-t-quantilizers-also-cut-off-the-upper-end-of-the",0,"",""],["A strong mind continues its trajectory of creativity","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yamkYG5dSzGcEdFTf/a-strong-mind-continues-its-trajectory-of-creativity",0,"",""],["Asking for online calls on AI s-risks discussions","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ahFdna5XAMxyuTssp/asking-for-online-calls-on-ai-s-risks-discussions",0,"",""],["CEA Should Invest in Helping Altruists Navigate Advanced AI","Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/946ymfxwd7YAC9yvT/cea-should-invest-in-helping-altruists-navigate-advanced-ai",0,"","forecasting"],["Difficulties in making powerful aligned AI","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nwjtqoox7JAcvMynx/difficulties-in-making-powerful-aligned-ai",0,"",""],["How much do markets value Open AI?","Ben_West","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZtZmkgDW6MH8AEEK6/how-much-do-markets-value-open-ai",0,"","forecasting"],["Simpler explanations of AGI risk","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/sjqGXmWdJWrRw8hBN/simpler-explanations-of-agi-risk",0,"",""],["A Study of AI Science Models","Eleni_A and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fTvw6K3CfxXdxAE5G/a-study-of-ai-science-models",0,"",""],["A Study of AI Science Models","Eleni Angelou and machinebiology","2023","blog","LessWrong","www.lesswrong.com/posts/h6TefRhwy6ioZrqXw/a-study-of-ai-science-models",0,"",""],["Are there enough opportunities for AI safety specialists?","mhint199","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ciwf4JXfMjqqz7oFn/are-there-enough-opportunities-for-ai-safety-specialists",0,"",""],["Can AI solve climate change?","Vivian","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xzJ9uNotWGDHznGi9/can-ai-solve-climate-change",0,"",""],["PCAST Working Group on Generative AI Invites Public Input","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/tC9NnWHNMN3wSNuXB/pcast-working-group-on-generative-ai-invites-public-input",0,"","governance"],["Steering GPT-2-XL by adding an activation vector","TurnTrout and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5spBue2z2tw4JuDCx/steering-gpt-2-xl-by-adding-an-activation-vector",0,"","interpretability"],["Aggregating Utilities for Corrigible AI [Feedback Draft]","Dan H and Simon Goldstein","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/k8KJqXyctf4a342QA/aggregating-utilities-for-corrigible-ai-feedback-draft",0,"",""],["Aggregating Utilities for Corrigible AI [Feedback Draft]","Dan H and Simon Goldstein","2023","blog","LessWrong","www.lesswrong.com/posts/k8KJqXyctf4a342QA/aggregating-utilities-for-corrigible-ai-feedback-draft",0,"",""],["Infinite-width MLPs as an \"ensemble prior\"","Vivek Hebbar","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CQMhLujqMpQ78Ru3R/infinite-width-mlps-as-an-ensemble-prior",0,"",""],["Input Swap Graphs: Discovering the role of neural network components at scale","Alexandre Variengien","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZSYo97kcfwtFdpcwe/input-swap-graphs-discovering-the-role-of-neural-network",0,"","interpretability"],["Towards Measures of Optimisation","mattmacdermott and Alexander Gietelink Oldenziel","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/X6ZjFShxNBNM5QCg4/towards-measures-of-optimisation-3",0,"","agents theory"],["Turning off lights with model editing","Sam Marks","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iNaaBAEkAy9nAgs3o/turning-off-lights-with-model-editing",0,"",""],["US public opinion of AI policy and risk","Jamie Elsey and David_Moss","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ConFiY9cRmg37fs2p/us-public-opinion-of-ai-policy-and-risk",0,"","policy"],["🏜️ EA is in Albuquerque!","Alex Long","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xkmiLmecWnD4LKRQ2/ea-is-in-albuquerque",0,"",""],["A more grounded idea of AI risk","Iknownothing","2023","blog","LessWrong","www.lesswrong.com/posts/ePKCgzzb3HhcnDwfr/a-more-grounded-idea-of-ai-risk",0,"",""],["A request to keep pessimistic AI posts actionable.","tcelferact","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ixua7wT7ZwuGfSLLi/a-request-to-keep-pessimistic-ai-posts-actionable-1",0,"",""],["Alignment, Goals, & The Gut-Head Gap: A Review of Ngo. et al","Violet Hour","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2yyZqRParGeLEja5u/alignment-goals-and-the-gut-head-gap-a-review-of-ngo-et-al",0,"",""],["Is Infra-Bayesianism Applicable to Value Learning?","RogerDearnaley","2023","blog","LessWrong","www.lesswrong.com/posts/j2Yi4WJgNWYKawweY/is-infra-bayesianism-applicable-to-value-learning",0,"",""],["Notes on the importance and implementation of safety-first cognitive architectures for AI","Brendon_Wong","2023","blog","LessWrong","www.lesswrong.com/posts/WLfogxQu4rDmone3z/notes-on-the-importance-and-implementation-of-safety-first",0,"","governance"],["A Corrigibility Metaphore - Big Gambles","WCargo","2023","blog","LessWrong","www.lesswrong.com/posts/pwPo7L2RjNvYDAo7g/a-corrigibility-metaphore-big-gambles",0,"",""],["AGI-Automated Interpretability is Suicide","__RicG__","2023","blog","LessWrong","www.lesswrong.com/posts/pQqoTTAnEePRDmZN4/agi-automated-interpretability-is-suicide",0,"","interpretability forecasting"],["AI interpretability could be harmful?","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/CRrkKAafopCmhJEBt/ai-interpretability-could-be-harmful",0,"","interpretability"],["Continuous doesn’t mean slow","Tom_Davidson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pR35WbLmruKdiMn2r/continuous-doesn-t-mean-slow",0,"","forecasting"],["Crises Reveal Centralisation (Stefan Schubert)","Will Howard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xczCcEhp4uy3zvNEv/crises-reveal-centralisation-stefan-schubert",0,"","policy"],["How much of a concern are open-source LLMs in the short, medium and long terms?","JavierCC","2023","blog","LessWrong","www.lesswrong.com/posts/EagMvzghoLue7yX4B/how-much-of-a-concern-are-open-source-llms-in-the-short",0,"","governance"],["New OpenAI Paper - Language models can explain neurons in language models","ViktorThink","2023","blog","LessWrong","www.lesswrong.com/posts/2qTxffyqeR4gbpEua/new-openai-paper-language-models-can-explain-neurons-in",0,"","interpretability"],["Roadmap for a collaborative prototype of an Open Agency Architecture","Deger Turan","2023","blog","LessWrong","www.lesswrong.com/posts/pHJtLHcWvfGbsW7LR/roadmap-for-a-collaborative-prototype-of-an-open-agency",0,"","governance"],["You don't need to be a genius to be in AI safety research","Claire Short","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kcE93PGPByM3Z7iGT/you-don-t-need-to-be-a-genius-to-be-in-ai-safety-research",0,"",""],["A note of caution on believing things on a gut level","Nathan_Barnard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CgeDuvedjqCj56HXZ/a-note-of-caution-on-believing-things-on-a-gut-level",0,"",""],["A Search for More ChatGPT / GPT-3.5 / GPT-4 \"Unspeakable\" Glitch Tokens","Martin Fell","2023","blog","LessWrong","www.lesswrong.com/posts/kmWrwtGE9B9hpbgRT/a-search-for-more-chatgpt-gpt-3-5-gpt-4-unspeakable-glitch",0,"","interpretability robustness"],["AI Safety Newsletter #5: Geoffrey Hinton speaks out on AI risk, the White House meets with AI labs, and Trojan attacks on language models","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CfFpEoibJTrTmiWtF/ai-safety-newsletter-5-geoffrey-hinton-speaks-out-on-ai-risk",0,"","governance policy"],["AI Safety Newsletter #5: Geoffrey Hinton speaks out on AI risk, the White House meets with AI labs, and Trojan attacks on language models","Dan H and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/X8yhso2noTKXfFE8s/ai-safety-newsletter-5-geoffrey-hinton-speaks-out-on-ai-risk",0,"",""],["Announcing “Key Phenomena in AI Risk” (facilitated reading group)","Nora_Ammann and particlemania","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mqvxR9nrXAzRr3ow9/announcing-key-phenomena-in-ai-risk-facilitated-reading",0,"",""],["Announcing “Key Phenomena in AI Risk” (facilitated reading group)","nora and particlemania","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WmnAQ4qTYwCviwDhS/announcing-key-phenomena-in-ai-risk-facilitated-reading",0,"",""],["Chilean AIS Hackathon Retrospective","Agustín Covarrubias and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/x9pcT6dvaGKox4PT5/chilean-ais-hackathon-retrospective",0,"",""],["Language models can explain neurons in language models","Steven Bills and 9 others","2023","report","openaipublic.blob.core.windows.net","openaipublic.blob.core.windows.net/neuron-explainer/paper/index.html",0,"",""],["Result Of The Bounty/Contest To Explain Infra-Bayes In The Language Of Game Theory","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/XCDwfr8LER8XnDaCv/result-of-the-bounty-contest-to-explain-infra-bayes-in-the",0,"",""],["Solving the Mechanistic Interpretability challenges: EIS VII Challenge 1","StefanHex and Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sTe78dNJDGywu9Dz6/solving-the-mechanistic-interpretability-challenges-eis-vii",0,"","interpretability mechanistic-interpretability"],["Stampy's AI Safety Info - New Distillations #2 [April 2023]","markov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FyfBJJJAZAdpfE9MX/stampy-s-ai-safety-info-new-distillations-2-april-2023",0,"",""],["Stopping dangerous AI: Ideal lab behavior","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/QJQEwcjp9zAr3bui2/stopping-dangerous-ai-ideal-lab-behavior",0,"","governance"],["Stopping dangerous AI: Ideal US behavior","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/Ybw7LfZWPRbEEKa5s/stopping-dangerous-ai-ideal-us-behavior",0,"","governance"],["When is Goodhart catastrophic?","Drake Thomas and Thomas Kwa","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fuSaKr6t6Zuh6GKaQ/when-is-goodhart-catastrophic",0,"","goodharts-law"],["Why \"just make an agent which cares only about binary rewards\" doesn't work.","Lysandre Terrisse","2023","blog","EA Forum","forum.effectivealtruism.org/posts/htgEGY5xbhFeJvt7E/why-just-make-an-agent-which-cares-only-about-binary-rewards",0,"","agents"],["A technical note on bilinear layers for interpretability","Lee Sharkey","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qm6bhbJmft2LJNzKH/a-technical-note-on-bilinear-layers-for-interpretability",0,"","interpretability"],["Acausal trade naturally results in the Nash bargaining solution","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/8oqzGF2p8H7JhFCdY/acausal-trade-naturally-results-in-the-nash-bargaining",0,"","theory"],["All AGI Safety questions welcome (especially basic ones) [May 2023]","steven0461","2023","blog","LessWrong","www.lesswrong.com/posts/SFuLQA7guCnG8pQ7T/all-agi-safety-questions-welcome-especially-basic-ones-may",0,"",""],["Annotated reply to Bengio's \"AI Scientists: Safe and Useful AI?\"","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/kGrwufqxfsyuaMREy/annotated-reply-to-bengio-s-ai-scientists-safe-and-useful-ai",0,"","automated-alignment-research"],["H-JEPA might be technically alignable in a modified form","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/umsGb5qkfzD3WarTR/h-jepa-might-be-technically-alignable-in-a-modified-form",0,"",""],["How \"AGI\" could end up being many different specialized AI's stitched together","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nAFavriWTLzmqTCcJ/how-agi-could-end-up-being-many-different-specialized-ai-s",0,"",""],["How quickly AI could transform the world (Tom Davidson on The 80,000 Hours Podcast)","80000_Hours and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/D8GitXAMt7deG8tBc/how-quickly-ai-could-transform-the-world-tom-davidson-on-the",0,"","forecasting"],["How The EthiSizer Almost Broke `Story'","Velikovsky_of_Newcastle","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iQCbubkxFCcZXmXZ9/how-the-ethisizer-almost-broke-story",0,"",""],["Import AI 328: Cheaper StableDiffusion; sim2soccer; AI refinement","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-328-cheaper-stablediffusion",0,"",""],["Inference Speed is Not Unbounded","OneManyNone","2023","blog","LessWrong","www.lesswrong.com/posts/Qvec2Qfm5H4WfoS9t/inference-speed-is-not-unbounded",0,"",""],["Is EDT correct? Does \"EDT\" == \"logical EDT\" == \"logical CDT\"?","Vivek Hebbar","2023","blog","LessWrong","www.lesswrong.com/posts/kowKm25hFxRquEyim/is-edt-correct-does-edt-logical-edt-logical-cdt",0,"","theory"],["LeCun’s “A Path Towards Autonomous Machine Intelligence” has an unsolved technical alignment problem","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/C5guLAx7ieQoowv3d/lecun-s-a-path-towards-autonomous-machine-intelligence-has-1",0,"",""],["Predictable updating about AI risk","Joe_Carlsmith","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3KAuAS2shyDwnjzNa/predictable-updating-about-ai-risk",0,"","forecasting"],["Reminder: AI Worldviews Contest Closes May 31","Jason Schukraft","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rvMirJZGLePztjHp8/reminder-ai-worldviews-contest-closes-may-31",0,"",""],["Unveiling the American Public Opinion on AI Moratorium and Government Intervention: The Impact of Media Exposure","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EoqeJCBiuJbMTKfPZ/unveiling-the-american-public-opinion-on-ai-moratorium-and",0,"","governance forecasting"],["What does it take to ban a thing?","qbolec","2023","blog","LessWrong","www.lesswrong.com/posts/cvpSNy32rgNdqqQ9r/what-does-it-take-to-ban-a-thing",0,"","governance"],["Against sacrificing AI transparency for generality gains","Ape in the coat","2023","blog","LessWrong","www.lesswrong.com/posts/XBTzNv9MfjYFf3nGG/against-sacrificing-ai-transparency-for-generality-gains",0,"","interpretability"],["An anthropomorphic AI dilemma","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/L2h9nAtPqEFK6atSJ/an-anthropomorphic-ai-dilemma",0,"",""],["An artificially structured argument for expecting AGI ruin","Rob Bensinger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QzkTfj4HGpLEdNjXX/an-artificially-structured-argument-for-expecting-agi-ruin",0,"",""],["Corrigibility, Much more detail than anyone wants to Read","Logan Zoellner","2023","blog","LessWrong","www.lesswrong.com/posts/v3jocJRScqkBGtwvf/corrigibility-much-more-detail-than-anyone-wants-to-read",0,"",""],["Graphical Representations of Paul Christiano's Doom Model","Nathan Young","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7p6CFnd6fYYqsH42r/graphical-representations-of-paul-christiano-s-doom-model",0,"","forecasting"],["Implications of the Whitehouse meeting with AI CEOs for AI superintelligence risk - a first-step towards evals?","Jamie Bernardi","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nTALzRAWxRnrxvoep/implications-of-the-whitehouse-meeting-with-ai-ceos-for-ai",0,"","evals governance policy"],["On the Loebner Silver Prize (a Turing test)","hold_my_fish","2023","blog","LessWrong","www.lesswrong.com/posts/mNR3tQDCxWL9XWv5u/on-the-loebner-silver-prize-a-turing-test",0,"","forecasting"],["Residual stream norms grow exponentially over the forward pass","StefanHex and TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8mizBCm3dyc432nK8/residual-stream-norms-grow-exponentially-over-the-forward",0,"","interpretability"],["Alignment as Function Fitting","A.H.","2023","blog","LessWrong","www.lesswrong.com/posts/YbGJqWNwwKsEDrHcf/alignment-as-function-fitting",0,"","alignment-faking deception"],["How much do you believe your results?","Eric Neyman","2023","blog","LessWrong","www.lesswrong.com/posts/nnDTgmzRrzDMiPF9B/how-much-do-you-believe-your-results",0,"","goodharts-law"],["Is \"red\" for GPT-4 the same as \"red\" for you?","Yusuke Hayashi","2023","blog","LessWrong","www.lesswrong.com/posts/LYgJrBf6awsqFRCt3/is-red-for-gpt-4-the-same-as-red-for-you",0,"",""],["My preferred framings for reward misspecification and goal misgeneralisation","Yi-Yang","2023","blog","LessWrong","www.lesswrong.com/posts/uxumdiuip7oWCAgaA/my-preferred-framings-for-reward-misspecification-and-goal",0,"",""],["Rank best universities for AI Saftey","Parker_Whitfill","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JFy5YewFngRgSJ3fd/rank-best-universities-for-ai-saftey",0,"",""],["An Update On The Campaign For AI Safety Dot Org","anonymous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zJYMkxGgpG8mCqagc/an-update-on-the-campaign-for-ai-safety-dot-org",0,"",""],["Intro to ML Safety virtual program: 12 June - 14 August","james and Oliver Z","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uB8BgEvvu5YXerFbw/intro-to-ml-safety-virtual-program-12-june-14-august-1",0,"",""],["Introducing the AI Objectives Institute's Research: Differential Paths toward Safe and Beneficial AI","cmck and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jra9axWurjMMYqxR5/introducing-the-ai-objectives-institute-s-research",0,"","governance"],["Orthogonal's","Tamsin Leake","2023","blog","carado.moe","carado.moe/formal-alignment-theory-change.html",0,"",""],["Orthogonal's Formal-Goal Alignment theory of change","Tamsin Leake","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4XcADCLDDguyej2N7/orthogonal-s-formal-goal-alignment-theory-of-change",0,"",""],["Regulate or Compete? The China Factor in U.S. AI Policy (NAIR #2)","charles_m","2023","blog","LessWrong","www.lesswrong.com/posts/DZzk8tLqbSCN5qe5M/regulate-or-compete-the-china-factor-in-u-s-ai-policy-nair-2",0,"","governance policy"],["Transcript of a presentation on catastrophic risks from AI","RobertM","2023","blog","LessWrong","www.lesswrong.com/posts/qoG4tR8TGEYjoDmw2/transcript-of-a-presentation-on-catastrophic-risks-from-ai",0,"",""],["[Link Post: New York Times] White House Unveils Initiatives to Reduce Risks of A.I.","Rockwell","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Cre2YC3hd5DeYLqDH/link-post-new-york-times-white-house-unveils-initiatives-to",0,"","governance"],["AI risk/reward: A simple model","Nathan Young","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MSkxRv8hviGvGgasD/ai-risk-reward-a-simple-model",0,"","forecasting"],["AI X-risk in the News: How Effective are Recent Media Items and How is Awareness Changing? Our New Survey Results.","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YweBjDwgdco669H72/ai-x-risk-in-the-news-how-effective-are-recent-media-items",0,"",""],["Clarifying and predicting AGI","Richard_Ngo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BoA3agdkAzL6HQtQP/clarifying-and-predicting-agi",0,"","forecasting"],["Most Leading AI Experts Believe That Advanced AI Could Be Extremely Dangerous to Humanity","jai","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fGfXrbtBJJasA2EKj/most-leading-ai-experts-believe-that-advanced-ai-could-be",0,"",""],["Trying to measure AI deception capabilities using temporary simulation fine-tuning","alenoach","2023","blog","LessWrong","www.lesswrong.com/posts/mmPohumufQJmCLeh6/trying-to-measure-ai-deception-capabilities-using-temporary",0,"","alignment-faking deception"],["White House Announces \"New Actions to Promote Responsible AI Innovation\"","berglund","2023","blog","LessWrong","www.lesswrong.com/posts/yBJftSnHcKAQkngzb/white-house-announces-new-actions-to-promote-responsible-ai",0,"","governance"],["Alignment Research @ EleutherAI","Curtis Huebner","2023","blog","blog.eleuther.ai","blog.eleuther.ai/alignment-eleuther/",0,"",""],["Finding Neurons in a Haystack: Case Studies with Sparse Probing","wesg and Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXiu6DBxKKWXC8Ygx/finding-neurons-in-a-haystack-case-studies-with-sparse",0,"","interpretability mechanistic-interpretability"],["How CISA can Support the Security of Large AI Models Against Theft [Grad School Assignment]","Harrison Durland","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wqZiSGi8effcRgiyh/how-cisa-can-support-the-security-of-large-ai-models-against",0,"","governance policy"],["How much do personal biases in risk assessment affect assessment of AI risks?","Gordon Seidoh Worley","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RzGJB7GoosQmfALuE/how-much-do-personal-biases-in-risk-assessment-affect",0,"",""],["My choice of AI misalignment introduction for a general audience","Bill","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Qq8GEy6beot2o2CWw/my-choice-of-ai-misalignment-introduction-for-a-general",0,"",""],["Prizes for matrix completion problems","paulfchristiano","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pJrebDRBj9gfBE8qE/prizes-for-matrix-completion-problems",0,"",""],["«Boundaries/Membranes» and AI safety compilation","Chipmonk","2023","blog","LessWrong","www.lesswrong.com/posts/fjgoMaBenyXcRDrbX/boundaries-membranes-and-ai-safety-compilation",0,"","theory"],["A Case for the Least Forgiving Take On Alignment","Thane Ruthenis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3JRBqRtHBDyPE3sGa/a-case-for-the-least-forgiving-take-on-alignment",0,"",""],["AGI rising: why we are in a new era of acute risk and increasing public awareness, and what to do now","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8YXFaM9yHbhiJTPqp/agi-rising-why-we-are-in-a-new-era-of-acute-risk-and",0,"","governance compute-governance"],["AGI safety career advice","Richard_Ngo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ho63vCb2MNFijinzY/agi-safety-career-advice",0,"",""],["AI Safety Newsletter #4: AI and Cybersecurity, Persuasive AIs, Weaponization, and Geoffrey Hinton talks AI risks","Center for AI Safety and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Ksqero4BmGFs8qfiC/ai-safety-newsletter-4-ai-and-cybersecurity-persuasive-ais",0,"",""],["AI Safety Newsletter #4: AI and Cybersecurity, Persuasive AIs, Weaponization, and Geoffrey Hinton talks AI risks","ozhang and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/MSw5y88tDbyp8WKo9/ai-safety-newsletter-4-ai-and-cybersecurity-persuasive-ais",0,"",""],["An Impossibility Proof Relevant to the Shutdown Problem and Corrigibility","Audere","2023","blog","LessWrong","www.lesswrong.com/posts/MBemd8k9uHFDEKzad/an-impossibility-proof-relevant-to-the-shutdown-problem-and",0,"","agents theory"],["Avoiding xrisk from AI doesn't mean focusing on AI xrisk","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Wigdk6s4xsC8fqhYT/avoiding-xrisk-from-ai-doesn-t-mean-focusing-on-ai-xrisk",0,"",""],["AXRP Episode 21 - Interpretability for Engineers with Stephen Casper","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9mzRsm5GyMYwvGSZi/axrp-episode-21-interpretability-for-engineers-with-stephen",0,"","interpretability"],["Finding Neurons in a Haystack: Case Studies with Sparse Probing","Wes Gurnee","2023","paper","arXiv preprint","arxiv.org/abs/2305.01610",0,"","interpretability mechanistic-interpretability"],["Owain Evans on LLMs, Truthful AI, AI Composition, and More","Ozzie Gooen and Owain_Evans","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aBKTkqeMC4AHoinFc/owain-evans-on-llms-truthful-ai-ai-composition-and-more",0,"","forecasting"],["P(doom|AGI) is high: why the default outcome of AGI is doom","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/THogLaytmj3n8oGbD/p-doom-or-agi-is-high-why-the-default-outcome-of-agi-is-doom",0,"","governance forecasting"],["Simulating a possible alignment solution in GPT2-medium using Archetypal Transfer Learning","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jE844jDBytBK8dWhw/simulating-a-possible-alignment-solution-in-gpt2-medium",0,"",""],["Systems that cannot be unsafe cannot be safe","Davidmanheim","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jiXMZHGmEf7qPrKPc/systems-that-cannot-be-unsafe-cannot-be-safe",0,"",""],["[Linkpost] ‘The Godfather of A.I.’ Leaves Google and Warns of Danger Ahead","Darius1","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pPQ5wqEPxLexCqGkL/linkpost-the-godfather-of-a-i-leaves-google-and-warns-of",0,"",""],["Call for Pythia-style foundation model suite for alignment research","Lucretia","2023","blog","EA Forum","forum.effectivealtruism.org/posts/P7x4hbanGKE2adfxe/call-for-pythia-style-foundation-model-suite-for-alignment",0,"","interpretability"],["CHAI Newsletter #1 2023","CHAI","2023","report","drive.google.com","drive.google.com/file/d/1aqZBYZFghWoZKjqZtSjfrkBelYFEglS4/view?usp=sharing",0,"",""],["Exploring Metaculus’s AI Track Record","Peter Scoblic and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/e9htD7txe8RDdcehm/exploring-metaculus-s-ai-track-record",0,"","forecasting"],["Global computing capacity","Vasco Grilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/L3mLPmBcsoXv36yt9/global-computing-capacity",0,"",""],["Import AI 327: Stable Diffusion on phones; GPT-Hacker; UK launches a £100m AI taskforce","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-327-stable-diffusion-on",0,"",""],["List of AI safety newsletters and other resources","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hsmh4fD8Dbkzvdehk/list-of-ai-safety-newsletters-and-other-resources",0,"",""],["My current take on existential AI risk [FB post]","Aryeh Englander","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uT2S5jWGEEi58bqby/my-current-take-on-existential-ai-risk-fb-post",0,"",""],["Retrospective on recent activity of Riesgos Catastróficos Globales","Jaime Sevilla","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7mSqokBNuHu3rzy4L/retrospective-on-recent-activity-of-riesgos-catastroficos",0,"",""],["Safety standards: a framework for AI regulation","joshc","2023","blog","LessWrong","www.lesswrong.com/posts/j2LD87wT3dpr4m7rs/safety-standards-a-framework-for-ai-regulation",0,"","governance"],["Shah (DeepMind) and Leahy (Conjecture) Discuss Alignment Cruxes","Olivia Jimenez and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cnn3kkC6kDqRkLe7W/shah-deepmind-and-leahy-conjecture-discuss-alignment-cruxes",0,"",""],["The costs of caution","Kelsey Piper","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bB2CSnFS6mEcNmPgD/the-costs-of-caution",0,"",""],["What 2025 looks like","Ruby","2023","blog","LessWrong","www.lesswrong.com/posts/JeMGZNZ6tuBWJHqvi/what-2025-looks-like",0,"","forecasting"],["A small update to the Sparse Coding interim research report","Lee Sharkey and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DezghAd4bdxivEknM/a-small-update-to-the-sparse-coding-interim-research-report",0,"","interpretability"],["Call for submissions: Choice of Futures survey questions","c.trout","2023","blog","LessWrong","www.lesswrong.com/posts/vnrazshJpxHbL6cdR/call-for-submissions-choice-of-futures-survey-questions",0,"","governance"],["Career uncertainty: Medicine vs. AI","MarkusK","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rhhgbuYBzkKkwjHhR/career-uncertainty-medicine-vs-ai",0,"",""],["Connectomics seems great from an AI x-risk perspective","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ybmDkJAj3rdrrauuu/connectomics-seems-great-from-an-ai-x-risk-perspective",0,"",""],["Discussion about AI Safety funding (FB transcript)","Akash","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eibgQcbRXtW7tukfv/discussion-about-ai-safety-funding-fb-transcript",0,"",""],["How does GPT-2 compute greater-than?: Interpreting mathematical abilities in a pre-trained language model","Michael Hanna","2023","paper","arXiv preprint","arxiv.org/abs/2305.00586",0,"","interpretability mechanistic-interpretability"],["The voyage of novelty","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aFZju8Kh4MWJaChav/the-voyage-of-novelty",0,"",""],["[SEE EDIT] No, *You* Need to Write Clearer","NicholasKross","2023","blog","LessWrong","www.lesswrong.com/posts/mLubC65xXekk5tkug/see-edit-no-you-need-to-write-clearer",0,"","interpretability"],["A Guide to Forecasting AI Science Capabilities","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/a6d78dLK8uESyjHEf/a-guide-to-forecasting-ai-science-capabilities-1",0,"","forecasting"],["A Guide to Forecasting AI Science Capabilities","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/KPqSFHdmGgfgznPvY/a-guide-to-forecasting-ai-science-capabilities",0,"","forecasting"],["Research agenda: Supervising AIs improving AIs","Quintin Pope and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/7e5tyFnpzGCdfT4mR/research-agenda-supervising-ais-improving-ais",0,"",""],["AI safety logo design contest, due end of May (extended)","Adrian Cipriani","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RPpAYgDF2NEFuvgwg/ai-safety-logo-design-contest-due-end-of-may-extended",0,"",""],["New open letter on AI — \"Include Consciousness Research\"","Jamie_Harris","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pZmjeb5RddWqsjp2j/new-open-letter-on-ai-include-consciousness-research",0,"",""],["The Social Alignment Problem","irving","2023","blog","LessWrong","www.lesswrong.com/posts/ecuAGrZGtavgsPs4w/the-social-alignment-problem",0,"",""],["Towards Automated Circuit Discovery for Mechanistic Interpretability","Arthur Conmy","2023","paper","arXiv preprint","arxiv.org/abs/2304.14997",0,"","interpretability mechanistic-interpretability"],["AI doom from an LLM-plateau-ist perspective","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KJRBb43nDxk6mwLcR/ai-doom-from-an-llm-plateau-ist-perspective",0,"",""],["How come there isn't that much focus in EA on research into whether / when AI's are likely to be sentient?","callum","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JZEgmumeamzBAAprt/how-come-there-isn-t-that-much-focus-in-ea-on-research-into",0,"",""],["Infrafunctions and Robust Optimization","Diffractor","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/d96dDEYMfnN2St3Bj/infrafunctions-and-robust-optimization",0,"",""],["Proposals for the AI Regulatory Sandbox in Spain","Guillem Bas and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6CibzfFnRWXcZosxv/proposals-for-the-ai-regulatory-sandbox-in-spain",0,"","evals governance policy"],["The AI guide I'm sending my grandparents","James Martin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uWgQv2gigQvDhQum6/the-ai-guide-i-m-sending-my-grandparents",0,"","policy"],["What are the limits of superintelligence?","rainy","2023","blog","LessWrong","www.lesswrong.com/posts/4f7cB6HKMT26N5t9b/what-are-the-limits-of-superintelligence",0,"","forecasting"],["A simple presentation of AI risk arguments","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/mJqabqwAb3QzZcu9T/a-simple-presentation-of-ai-risk-arguments",0,"",""],["A very non-technical explanation of the basics of infra-Bayesianism","matolcsid","2023","blog","LessWrong","www.lesswrong.com/posts/NdJsWDS7Aq4xqoumk/a-very-non-technical-explanation-of-the-basics-of-infra",0,"","agents theory"],["Archetypal Transfer Learning: a Proposed Alignment Solution that solves the Inner & Outer Alignment Problem while adding Corrigible Traits to GPT-2-medium","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/pu6D2EdJiz2mmhxfB/archetypal-transfer-learning-a-proposed-alignment-solution",0,"",""],["Archetypal Transfer Learning: a Proposed Alignment Solution that solves the Inner x Outer Alignment Problem while adding Corrigible Traits to GPT-2-medium","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XasETzmipXj4dgz7e/archetypal-transfer-learning-a-proposed-alignment-solution",0,"","evals"],["How Many Bits Of Optimization Can One Bit Of Observation Unlock?","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6rZroG3mztowktJzp/how-many-bits-of-optimization-can-one-bit-of-observation",0,"",""],["I was Wrong, Simulator Theory is Real","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/mfn32QHwKb55afHq4/i-was-wrong-simulator-theory-is-real",0,"","interpretability deception"],["Infra-Bayesianism naturally leads to the monotonicity principle, and I think this is a problem","matolcsid","2023","blog","LessWrong","www.lesswrong.com/posts/yykNvq257zBLDNmJo/infra-bayesianism-naturally-leads-to-the-monotonicity",0,"","agents theory"],["Is there EA discussion on non-x-risk transformative AI?","Franziska Fischer","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jBYmcCe9aarTLs8nf/is-there-ea-discussion-on-non-x-risk-transformative-ai",0,"",""],["Join a ‘learning by writing' group","jwpieters","2023","blog","EA Forum","forum.effectivealtruism.org/posts/udwe7LDvRFrhp5dFD/join-a-learning-by-writing-group",0,"",""],["LM Situational Awareness, Evaluation Proposal: Violating Imitation","Jacob Pfau","2023","blog","LessWrong","www.lesswrong.com/posts/kkaBC9Epydj3m6ZsA/lm-situational-awareness-evaluation-proposal-violating",0,"","evals deception situational-awareness"],["AI Safety Newsletter #3: AI policy proposals and a new challenger approaches","Oliver Z and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qy25pydHAYZoCFsAG/ai-safety-newsletter-3-ai-policy-proposals-and-a-new",0,"","governance policy"],["Briefly how I've updated since ChatGPT","rime","2023","blog","LessWrong","www.lesswrong.com/posts/3DyXQkkkGnSgy95ex/briefly-how-i-ve-updated-since-chatgpt",0,"","governance"],["Is China Becoming a Science and Technology Superpower? Jeffrey Ding's Insight on China's Diffusion Deficit","Wyman Kwok","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7nFw536oK9H8rZmCP/is-china-becoming-a-science-and-technology-superpower",0,"","policy"],["Making Nanobots isn't a one-shot process, even for an artificial superintelligance","dankrad","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fc9KjZeSLuHN7HfW6/making-nanobots-isn-t-a-one-shot-process-even-for-an",0,"",""],["My Assessment of the Chinese AI Safety Community","Lao Mein","2023","blog","LessWrong","www.lesswrong.com/posts/EAwe7smpmFQi2653G/my-assessment-of-the-chinese-ai-safety-community",0,"","governance"],["Notes on Potential Future AI Tax Policy","Zvi","2023","blog","LessWrong","www.lesswrong.com/posts/HMQmEcL3CTs3TBty7/notes-on-potential-future-ai-tax-policy",0,"","governance policy"],["Paths to failure","Karl von Wendt and mespa","2023","blog","LessWrong","www.lesswrong.com/posts/yv4xAnkEyWvpXNBte/paths-to-failure",0,"",""],["Reframing the burden of proof: Companies should prove that models are safe (rather than expecting auditors to prove that models are dangerous)","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/eJ8xrMeWMqQHEN2vm/reframing-the-burden-of-proof-companies-should-prove-that",0,"","evals governance"],["UK Government announces £100 million in funding for Foundation Model Taskforce.","jwpieters","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Knh38LJi5aQDkFKMc/uk-government-announces-gbp100-million-in-funding-for",0,"","policy"],["A concise sum-up of the basic argument for AI doom","Mergimio H. Doefevmil","2023","blog","LessWrong","www.lesswrong.com/posts/FDnLNvNqDifsviJyW/a-concise-sum-up-of-the-basic-argument-for-ai-doom",0,"",""],["Consequentialism is in the Stars not Ourselves","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/PtEPqonFDv7ueYYpu/consequentialism-is-in-the-stars-not-ourselves",0,"","agents theory"],["For alignment, we should simultaneously use multiple theories of cognition and value","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FnwqLB7A9PenRdg4Z/for-alignment-we-should-simultaneously-use-multiple-theories",0,"",""],["FT: We must slow down the race to God-like AI","Angelina Li","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FdmdsXuAbzwAjTecr/ft-we-must-slow-down-the-race-to-god-like-ai",0,"",""],["Import AI 326:Chinese AI regulations; Stability's new LMs If AI is fashionable in 2023, then what will be fashionable in 2024?","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-distributed-ai-chinese-ai",0,"","governance"],["No, the EMH does not imply that markets have long AGI timelines","Jakob","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Go5CDwyna3hAfngKP/no-the-emh-does-not-imply-that-markets-have-long-agi",0,"","governance forecasting"],["Student competition for drafting a treaty on moratorium of large-scale AI capabilities R&D","Nayanika","2023","blog","EA Forum","forum.effectivealtruism.org/posts/epTvpAEfCY74CMdMv/student-competition-for-drafting-a-treaty-on-moratorium-of",0,"","governance policy"],["The Case For Civil Disobedience For The AI Movement","Murali Thoppil","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JMb37qrCYCeKqFxtp/the-case-for-civil-disobedience-for-the-ai-movement",0,"",""],["Value Learning – Towards Resolving Confusion","PashaKamyshev","2023","blog","LessWrong","www.lesswrong.com/posts/QwrCB6dSSkFCyGbSG/value-learning-towards-resolving-confusion",0,"",""],["X-Risk Researchers Survey","NitaSangha","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7GvSxbkAgpMEHpuJJ/x-risk-researchers-survey",0,"","policy forecasting"],["A great talk for AI noobs (according to an AI noob)","Dov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/REjSEuKbzh2QRFBgK/a-great-talk-for-ai-noobs-according-to-an-ai-noob",0,"",""],["A great talk for AI noobs (according to an AI noob)","dov","2023","blog","LessWrong","www.lesswrong.com/posts/dWbhbpEMKysdYt89c/a-great-talk-for-ai-noobs-according-to-an-ai-noob",0,"",""],["Endo-, Dia-, Para-, and Ecto-systemic novelty","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zwRQW9gEyszmHwff8/endo-dia-para-and-ecto-systemic-novelty",0,"",""],["Preventing AI Misuse: State of the Art Research and its Flaws","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/u9atkEFDcuMkgipch/preventing-ai-misuse-state-of-the-art-research-and-its-flaws",0,"",""],["Why do we care about agency for alignment?","Chris_Leong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tNYfsw5q873jjXfCP/why-do-we-care-about-agency-for-alignment",0,"",""],["PhD Position: AI Interpretability in Berlin, Germany","Stephan_Wäldchen","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NkmxjzHbk5WxvK5xs/phd-position-ai-interpretability-in-berlin-germany",0,"","interpretability"],["The Security Mindset, S-Risk and Publishing Prosaic Alignment Research","marc/er","2023","blog","LessWrong","www.lesswrong.com/posts/dRAmQrvXAnwLEsFzv/the-security-mindset-s-risk-and-publishing-prosaic-alignment",0,"",""],["\"Who Will You Be After ChatGPT Takes Your Job?\"","Stephen Thomas","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8kKLSy285eWbFn4qC/who-will-you-be-after-chatgpt-takes-your-job",0,"",""],["AI Progress: The Game Show","Alex Arnett","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cgxSATRxfn9X9H8rd/ai-progress-the-game-show",0,"",""],["If your AGI x-risk estimates are low, what scenarios make up the bulk of your expectations for an OK outcome?","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/idjzaqfGguEAaC34j/if-your-agi-x-risk-estimates-are-low-what-scenarios-make-up",0,"",""],["OPEC for a slow AGI takeoff","vyrax","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nczavbHtYCjwrRK75/opec-for-a-slow-agi-takeoff",0,"","governance forecasting"],["Should we publish mechanistic interpretability research?","Marius Hobbhahn and LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iDNEjbdHhjzvLLAmm/should-we-publish-mechanistic-interpretability-research",0,"","interpretability mechanistic-interpretability"],["The basic reasons I expect AGI ruin","Rob Bensinger","2023","blog","intelligence.org","intelligence.org/2023/04/21/the-basic-reasons-i-expect-agi-ruin/",0,"",""],["the multiverse argument argument against automated alignment","Tamsin Leake","2023","blog","carado.moe","carado.moe/multiverse-argument-automated-alignment.html",0,"","automated-alignment-research"],["Thinking about maximization and corrigibility","James Payor","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/AFJgo99YckQnhbF8Z/thinking-about-maximization-and-corrigibility",0,"","goodharts-law"],["Alien Axiology","snerx","2023","blog","LessWrong","www.lesswrong.com/posts/wMQw3P8KmbCvNbN4j/alien-axiology",0,"","instrumental-convergence theory"],["An open letter to SERI MATS program organisers","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/bRtP7Mub3hXAoo4vQ/an-open-letter-to-seri-mats-program-organisers",0,"",""],["Behavioural statistics for a maze-solving agent","peligrietzer and TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/eowhY5NaCaqY6Pkj9/behavioural-statistics-for-a-maze-solving-agent",0,"","interpretability agents"],["Ideas for studies on AGI risk","dr_s","2023","blog","LessWrong","www.lesswrong.com/posts/76n4pMcoDBTdXHTLY/ideas-for-studies-on-agi-risk",0,"","instrumental-convergence power-seeking"],["Japan AI Alignment Conference Postmortem","Chris Scammell and Katrina Joslin","2023","blog","LessWrong","www.lesswrong.com/posts/Yc6cpGmBieS7ADxcS/japan-ai-alignment-conference-postmortem",0,"",""],["Language Models are a Potentially Safe Path to Human-Level AGI","Nadav Brandes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wNrbHbhgPJBD2d9v6/language-models-are-a-potentially-safe-path-to-human-level",0,"","interpretability chain-of-thought-faithfulness"],["Merger of DeepMind and Google Brain","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ELn2cdcDwAET6fqEi/merger-of-deepmind-and-google-brain",0,"","governance"],["OpenAI could help X-risk by wagering itself","VojtaKovarik","2023","blog","LessWrong","www.lesswrong.com/posts/aEF24JnXFhiR4bkEu/openai-could-help-x-risk-by-wagering-itself-1",0,"","governance"],["Proposal: Using Monte Carlo tree search instead of RLHF for alignment research","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/rPCHKfSYFSWGYpTKX/proposal-using-monte-carlo-tree-search-instead-of-rlhf-for",0,"","rlhf"],["Reasons to have hope","jwpieters","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pFW5dfCEFwuLcwfpk/reasons-to-have-hope",0,"",""],["Responsible Deployment in 20XX","Carson","2023","blog","LessWrong","www.lesswrong.com/posts/D7ig57J8tWEsMQwcx/responsible-deployment-in-20xx",0,"","evals governance"],["Stability AI releases StableLM, an open-source ChatGPT counterpart","Ozyrus","2023","blog","LessWrong","www.lesswrong.com/posts/LSFpWmrsvw32teiLB/stability-ai-releases-stablelm-an-open-source-chatgpt",0,"","forecasting"],["The Economist feature articles on LLMs","Dr Dan Epstein","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QsHH66kyN4GJhBqpK/the-economist-feature-articles-on-llms",0,"",""],["'AI Emergency Eject Criteria' Survey","tcelferact","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BFNxoQnqc9zmoB3wi/ai-emergency-eject-criteria-survey",0,"",""],["12 tentative ideas for US AI policy (Luke Muehlhauser)","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iiRGCydMX7aiEjvGm/12-tentative-ideas-for-us-ai-policy-luke-muehlhauser",0,"","evals governance policy"],["[Crosspost] Organizing a debate with experts and MPs to raise AI xrisk awareness: a possible blueprint","otto.barten","2023","blog","LessWrong","www.lesswrong.com/posts/rTgwxsxu6hstgxDR2/crosspost-organizing-a-debate-with-experts-and-mps-to-raise",0,"","governance"],["Approximation is expensive, but the lunch is cheap","Jesse Hoogland and Zach Furman","2023","blog","LessWrong","www.lesswrong.com/posts/gq9GR6duzcuxyxZtD/approximation-is-expensive-but-the-lunch-is-cheap",0,"","interpretability"],["Artificial Intelligence: Challenges and Opportunities for the Department of Defense","Jason Matheny and Rand Corporation","2023","report","rand.org","www.rand.org/content/dam/rand/pubs/testimonies/CTA2700/CTA2723-1/RAND_CTA2723-1.pdf",0,"",""],["Davidad's Bold Plan for Alignment: An In-Depth Explanation","Charbel-Raphaël and Gabin","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jRf4WENQnhssCb6mJ/davidad-s-bold-plan-for-alignment-an-in-depth-explanation",0,"","governance"],["Is there any literature on using socialization for AI alignment?","Nathan1123","2023","blog","LessWrong","www.lesswrong.com/posts/GrvYBp7c2wz4fryb2/is-there-any-literature-on-using-socialization-for-ai",0,"",""],["Organizing a debate with experts and MPs to raise AI xrisk awareness: a possible blueprint","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/J3ribNjvPRtHCK7bC/organizing-a-debate-with-experts-and-mps-to-raise-ai-xrisk",0,"",""],["Orthogonal: A new agent foundations alignment organization","Tamsin Leake","2023","blog","LessWrong","www.lesswrong.com/posts/b2xTk6BLJqJHd3ExE/orthogonal-a-new-agent-foundations-alignment-organization",0,"","agents theory"],["Paying the corrigibility tax","Max H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zwPsiWY7FJc8QkpDJ/paying-the-corrigibility-tax",0,"",""],["The Learning-Theoretic Agenda: Status 2023","Vanessa Kosoy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZwshvqiqCvXPsZEct/the-learning-theoretic-agenda-status-2023",0,"","agents theory"],["[Linkpost] AI Alignment, Explained in 5 Points (updated)","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DLEdzaiSqoC4eonKp/linkpost-ai-alignment-explained-in-5-points-updated",0,"",""],["AI Safety Newsletter #2: ChaosGPT, Natural Selection, and AI Safety in the Media","Oliver Z and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Eu4ZDCt2yaKavtQ9s/ai-safety-newsletter-2-chaosgpt-natural-selection-and-ai",0,"",""],["AI Safety Newsletter #2: ChaosGPT, Natural Selection, and AI Safety in the Media","ozhang and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/9CdBJZTKKZ5bEDJys/ai-safety-newsletter-2-chaosgpt-natural-selection-and-ai",0,"",""],["Capabilities and alignment of LLM cognitive architectures","Seth Herd","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ogHr8SvGqg9pW5wsT/capabilities-and-alignment-of-llm-cognitive-architectures",0,"","chain-of-thought-faithfulness"],["Risk of AI deceleration.","Micah Zoltu","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Z4tsromjxAbMpAtiZ/risk-of-ai-deceleration",0,"",""],["Scientism vs. people","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/3BPuuNDavJ2drKvGK/scientism-vs-people",0,"","automated-alignment-research governance"],["The Guardian Version 1","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/b56nedeCALDuvPxWB/the-guardian-version-1",0,"",""],["Transformer Math 101","Quentin Anthony and 2 others","2023","blog","blog.eleuther.ai","blog.eleuther.ai/transformer-math/",0,"",""],["What can we do now to prepare for AI sentience, in order to protect them from the global scale of human sadism?","rime","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2hwhxpFfjR3Bhf3Ya/what-can-we-do-now-to-prepare-for-ai-sentience-in-order-to",0,"",""],["AI Alignment Research Engineer Accelerator (ARENA): call for applicants","TheMcDouglas","2023","blog","LessWrong","www.lesswrong.com/posts/CNytdmT6xrWdexQgN/ai-alignment-research-engineer-accelerator-arena-call-for",0,"",""],["AI Impacts Quarterly Newsletter, Jan-Mar 2023","Harlan Stewart","2023","blog","aiimpacts.org","aiimpacts.org/ai-impacts-quarterly-newsletter-jan-mar-2023/",0,"",""],["AI Impacts Quarterly Newsletter, Jan-Mar 2023","Harlan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Qpgfrde9w9vCT3vgm/ai-impacts-quarterly-newsletter-jan-mar-2023",0,"","governance"],["AI policy ideas: Reading list","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/NfqqsHqembNEsTrSr/ai-policy-ideas-reading-list",0,"","governance policy"],["An alternative of PPO towards alignment","ml hkust","2023","blog","LessWrong","www.lesswrong.com/posts/iuorxZu6tLFhP7oQY/an-alternative-of-ppo-towards-alignment",0,"","rlhf"],["But why would the AI kill us?","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/87EzRDAHkQJptLthE/but-why-would-the-ai-kill-us",0,"",""],["Geoffrey Miller on Cross-Cultural Understanding Between China and Western Countries as a Neglected Consideration in AI Alignment","Evan_Gaensbauer","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wxjuboWosFP7ez5tM/geoffrey-miller-on-cross-cultural-understanding-between",0,"","governance"],["Import AI 325: Automated mad science; AI vs democracy; and a 12B parameter language model","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-325-automated-mad-science",0,"",""],["Prediction: any uncontrollable AI will turn earth into a giant computer","Karl von Wendt","2023","blog","LessWrong","www.lesswrong.com/posts/jkaLGoNLdsp654KhD/prediction-any-uncontrollable-ai-will-turn-earth-into-a",0,"",""],["Slowing AI: Foundations","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/MoLLqFtMup39PCsaG/slowing-ai-foundations",0,"","governance"],["What is your timelines for ADI (artificial disempowering intelligence)?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/hQes2GNPcck6WmrrP/what-is-your-timelines-for-adi-artificial-disempowering",0,"","forecasting"],["[Link/crosspost] [US] NTIA: AI Accountability Policy Request for Comment","Kyle J. Lucchese","2023","blog","LessWrong","www.lesswrong.com/posts/hGWYDxz9Pf8haQs49/link-crosspost-us-ntia-ai-accountability-policy-request-for",0,"","governance policy"],["Mechanistically interpreting time in GPT-2 small","rgould and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/6tHNM2s6SWzFHv3Wo/mechanistically-interpreting-time-in-gpt-2-small",0,"","interpretability"],["Possibilizing vs. actualizing","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5uyRB4CGAvB2eLHvm/possibilizing-vs-actualizing",0,"",""],["Summary: The Case for Halting AI Development - Max Tegmark on the Lex Fridman Podcast","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/akbwyBioGBd68CsNx/summary-the-case-for-halting-ai-development-max-tegmark-on",0,"","governance"],["An example elevator pitch for AI doom","laserfiche","2023","blog","LessWrong","www.lesswrong.com/posts/EEWzh3oDTpCNEkqzX/an-example-elevator-pitch-for-ai-doom",0,"",""],["Brain-computer interfaces and brain organoids in AI alignment?","freedomandutility","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4fTrqJ7w8weRCYHeF/brain-computer-interfaces-and-brain-organoids-in-ai",0,"",""],["Concave Utility Question","Scott Garrabrant","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uJnR4YmG5Kq9FfTey/concave-utility-question",0,"",""],["Concrete, existing examples of high-impact risks from AI?","freedomandutility","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2SW5cc9ZGmzAF8Xbf/concrete-existing-examples-of-high-impact-risks-from-ai",0,"",""],["FLI report: Policymaking in the Pause","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/ERsbTthnzWLmNDCb5/fli-report-policymaking-in-the-pause",0,"","governance policy"],["Open-source LLMs may prove Bostrom's vulnerable world hypothesis","Roope Ahvenharju","2023","blog","LessWrong","www.lesswrong.com/posts/BmTG3tiBnqyckA3LJ/open-source-llms-may-prove-bostrom-s-vulnerable-world",0,"","governance"],["SmartyHeaderCode: anomalous tokens for GPT3.5 and GPT-4","AdamYedidia","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ChtGdxk9mwZ2Rxogt/smartyheadercode-anomalous-tokens-for-gpt3-5-and-gpt-4-1",0,"","interpretability robustness"],["Who is testing AI Safety public outreach messaging?","anonymous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/s5AAzpmbqedKgEaDj/who-is-testing-ai-safety-public-outreach-messaging",0,"",""],["\"Do X because decision theory\" ~= \"Do X because bayes theorem\"","lc","2023","blog","LessWrong","www.lesswrong.com/posts/KDzykLCYMfWiRiWnd/do-x-because-decision-theory-do-x-because-bayes-theorem",0,"","theory"],["\"Risk Awareness Moments\" (Rams): A concept for thinking about AI governance interventions","oeg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EcrNFxGszfgcGevtf/risk-awareness-moments-rams-a-concept-for-thinking-about-ai",0,"","governance policy"],["[linkpost] \"What Are Reasonable AI Fears?\" by Robin Hanson, 2023-04-23","Arjun Panickssery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XnnfPC2gsgRFZezkE/linkpost-what-are-reasonable-ai-fears-by-robin-hanson-2023",0,"","governance forecasting"],["[Linkpost] The A.I. Dilemma - March 9, 2023, with Tristan Harris and Aza Raskin","PeterSlattery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DHJh3fuXK3TCtBsNq/linkpost-the-a-i-dilemma-march-9-2023-with-tristan-harris",0,"",""],["A freshman year during the AI midgame: my approach to the next year","Buck","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2DzLY6YP2z5zRDAGA/a-freshman-year-during-the-ai-midgame-my-approach-to-the",0,"","robustness"],["AI Safety Europe Retreat 2023 Retrospective","Magdalena Wache","2023","blog","LessWrong","www.lesswrong.com/posts/sjx9i6ndoNHwYg9N6/ai-safety-europe-retreat-2023-retrospective",0,"",""],["Anti-'FOOM' (stop trying to make your cute pet name the thing)","david_reinstein","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xsKwDuggxcYpYCe2z/anti-foom-stop-trying-to-make-your-cute-pet-name-the-thing",0,"",""],["GPT-4 is easily controlled/exploited with tricky decision theoretic dilemmas.","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/paYyQ8Y7Zun5ERRj3/gpt-4-is-easily-controlled-exploited-with-tricky-decision",0,"","theory"],["List of requests for an AI slowdown/halt.","Cleo Nardo","2023","blog","LessWrong","www.lesswrong.com/posts/a87uzervYEYJ8pCgk/list-of-requests-for-an-ai-slowdown-halt",0,"","governance"],["Prospects for AI safety agreements between countries","oeg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/L8GjzvRYA9g9ox2nP/prospects-for-ai-safety-agreements-between-countries",0,"","governance policy"],["Research Report: Incorrectness Cascades","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/M5vEjix8oPeWXeGFY/research-report-incorrectness-cascades",0,"","interpretability deception"],["Shapley Value Attribution in Chain of Thought","leogao","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FX5JmftqL2j6K8dn4/shapley-value-attribution-in-chain-of-thought",0,"","interpretability chain-of-thought-faithfulness"],["The self-unalignment problem","Jan_Kulveit and rosehadshar","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9GyniEBaN3YYTqZXn/the-self-unalignment-problem",0,"","agents"],["What we’ve learned so far from our technological temptations project","richardkorzekwa","2023","blog","aiimpacts.org","aiimpacts.org/what-weve-learned-so-far-from-our-technological-temptations-project/",0,"",""],["\"Aligned\" foundation models don't imply aligned systems","Max H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gmHiwafywFo33euGz/aligned-foundation-models-don-t-imply-aligned-systems",0,"",""],["[US] NTIA: AI Accountability Policy Request for Comment","Kyle J. Lucchese","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GcrKndFY2oSKEFLub/us-ntia-ai-accountability-policy-request-for-comment",0,"","governance policy"],["AGI - alignment - paperclip maximizer - pause - defection - incentives","Mars Robertson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nwEDksQ8GCCjBiMnF/agi-alignment-paperclip-maximizer-pause-defection-incentives",0,"",""],["Announcing Epoch’s dashboard of key trends and figures in Machine Learning","Jsevillamol","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/opsfYWNxBYF5sJujB/announcing-epoch-s-dashboard-of-key-trends-and-figures-in",0,"",""],["Financial Times: We must slow down the race to God-like AI","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/sj84MyKXZKZwqkCNh/financial-times-we-must-slow-down-the-race-to-god-like-ai",0,"","governance"],["Identifying semantic neurons, mechanistic circuits & interpretability web apps","Esben Kran and Neel Nanda","2023","blog","LessWrong","www.lesswrong.com/posts/sruT3a9KhyLnYmLi7/identifying-semantic-neurons-mechanistic-circuits-and",0,"","interpretability mechanistic-interpretability"],["Intro to Ontogenetic Curriculum","Eris","2023","blog","LessWrong","www.lesswrong.com/posts/RCbofC8fCJ6NnYti7/intro-to-ontogenetic-curriculum",0,"",""],["Navigating the Open-Source AI Landscape: Data, Funding, and Safety","André Ferretti and mic","2023","blog","LessWrong","www.lesswrong.com/posts/69CRFgqbQyFBoYcg5/navigating-the-open-source-ai-landscape-data-funding-and",0,"","governance"],["What is the best source to explain short AI timelines to a skeptical person?","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/6z4jkLkDpzMC4ffEW/what-is-the-best-source-to-explain-short-ai-timelines-to-a",0,"","forecasting"],["[Link] Sarah Constantin: \"Why I am Not An AI Doomer\"","lbThingrb","2023","blog","LessWrong","www.lesswrong.com/posts/nG4biq5ymbBviKYsJ/link-sarah-constantin-why-i-am-not-an-ai-doomer",0,"","forecasting"],["[linkpost] AI NOW Institute's 2023 Annual Report & Roadmap","Tristan Williams","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Bj4SBXtnjGmfH4QFq/linkpost-ai-now-institute-s-2023-annual-report-and-roadmap",0,"","governance"],["AGI goal space is big, but narrowing might not be as hard as it seems.","Jacy Reese Anthis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nxmo2cyREteqvLMss/agi-goal-space-is-big-but-narrowing-might-not-be-as-hard-as-1",0,"",""],["AI x-risk, approximately ordered by embarrassment","Alex Lawsen","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mSF4KTxAGRG3EHmhb/ai-x-risk-approximately-ordered-by-embarrassment",0,"","deception"],["AIs accelerating AI research","Ajeya","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hw8ePRLJop7kSEZK3/ais-accelerating-ai-research",0,"",""],["Alignment of AutoGPT agents","Ozyrus","2023","blog","LessWrong","www.lesswrong.com/posts/JnAh4YHfrYpPNwc8Y/alignment-of-autogpt-agents",0,"","agents forecasting chain-of-thought-faithfulness"],["Apply to >50 AI safety funders in one application with the Nonlinear Network [Round Closed]","Drew Spartz and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Qoecey2umNjcqEGHP/apply-to-greater-than-50-ai-safety-funders-in-one",0,"",""],["Artificial Intelligence as exit strategy from the age of acute existential risk","Arturo Macias","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6j6qgNa3uGmzJEMoN/artificial-intelligence-as-exit-strategy-from-the-age-of",0,"",""],["AXRP Episode 20 - ‘Reform’ AI Alignment with Scott Aaronson","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nsygJvidfgidmgKqX/axrp-episode-20-reform-ai-alignment-with-scott-aaronson",0,"",""],["Boundaries-based security and AI safety approaches","Allison Duettmann","2023","blog","LessWrong","www.lesswrong.com/posts/TZy4mFJFJ4yv2MRhg/boundaries-based-security-and-ai-safety-approaches",0,"",""],["Gradient Descent in Activation Space: a Tale of Two Papers","Blaine","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HHSuvG2hqAnGT5Wzp/gradient-descent-in-activation-space-a-tale-of-two-papers",0,"","interpretability"],["Localizing Model Behavior With Path Patching","","2023","paper","arXiv preprint","arxiv.org/abs/2304.05969",0,"","evals"],["Natural language alignment","Jacy Reese Anthis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EhkHnNJXwT8RmtfYZ/natural-language-alignment-1",0,"","rlhf alignment-faking deception"],["Navigating the Open-Source AI Landscape: Data, Funding, and Safety","AndreFerretti and mic","2023","blog","EA Forum","forum.effectivealtruism.org/posts/N25EARxvbxYJa5pbB/navigating-the-open-source-ai-landscape-data-funding-and",0,"","governance"],["Towards a solution to the alignment problem via objective detection and evaluation","Paul Colognese","2023","blog","LessWrong","www.lesswrong.com/posts/vZCSPffGLhJT3heqc/towards-a-solution-to-the-alignment-problem-via-objective",0,"","interpretability evals alignment-faking deception"],["[Linkpost] 538 Politics Podcast on AI risk & politics","jackva","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Sa4ahq8AGTniuuvjE/linkpost-538-politics-podcast-on-ai-risk-and-politics",0,"","policy"],["[MLSN #9] Verifying large training runs, security risks from LLM access to APIs, why natural selection may favor AIs over humans","Dan H and ThomasW","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2BG49yHpgEL46eioZ/mlsn-9-verifying-large-training-runs-security-risks-from-llm",0,"",""],["[MLSN #9] Verifying large training runs, security risks from LLM access to APIs, why natural selection may favor AIs over humans","ThomasW and Dan H","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9mFT7rm9wvmq9uB2m/mlsn-9-verifying-large-training-runs-security-risks-from-llm",0,"",""],["AI Risk US Presidental Candidate","Simon Berens","2023","blog","EA Forum","forum.effectivealtruism.org/posts/knCbv3LxwciWHfACE/ai-risk-us-presidental-candidate",0,"","policy"],["Cyberspace Administration of China: Draft of \"Regulation for Generative Artificial Intelligence Services\" is open for comments","sanxiyn","2023","blog","LessWrong","www.lesswrong.com/posts/yndw9qQFsNkXdTu3K/cyberspace-administration-of-china-draft-of-regulation-for",0,"","governance"],["Data Taxation: A Proposal for Slowing Down AGI Progress","Per Ivar Friborg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2w7fKv5EzZfrqm5Tg/data-taxation-a-proposal-for-slowing-down-agi-progress",0,"","governance policy"],["Evolution provides no evidence for the sharp left turn","Quintin Pope","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hvz9qjWyv8cLX9JJR/evolution-provides-no-evidence-for-the-sharp-left-turn",0,"","forecasting"],["Existential risk x Crypto: An unconference at Zuzalu","Yesh","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Cfse8PEW8yDC9oAdu/existential-risk-x-crypto-an-unconference-at-zuzalu",0,"",""],["FLI And Eliezer Should Reach Consensus","JenniferRM","2023","blog","LessWrong","www.lesswrong.com/posts/BWLKRMQn3DFcQg6of/fli-and-eliezer-should-reach-consensus",0,"","governance"],["Four mindset disagreements behind existential risk disagreements in ML","RobBensinger","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JQxvZZdPG5KYjyBfg/four-mindset-disagreements-behind-existential-risk",0,"",""],["Import AI 324: Machiavellian AIs; LLMs and political campaigns; Facebook makes an excellent segmentation model","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-324-machiavellian-ais-llms",0,"",""],["Introducing the Mental Health Roadmap Series","Emily and Dave Cortright","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iJRrZQofGvt6q5nYg/introducing-the-mental-health-roadmap-series",0,"",""],["Measuring artificial intelligence on human benchmarks is naive","Ward A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZjQ2fXpATBMvnBzzj/measuring-artificial-intelligence-on-human-benchmarks-is",0,"","benchmarks"],["Metaculus’ predictions are much better than low-information priors","Vasco Grilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JN6wm6u5MMmqwdnEs/metaculus-predictions-are-much-better-than-low-information",0,"","forecasting"],["ML Safety Newsletter #9","Dan Hendrycks","2023","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-9",0,"",""],["National Telecommunications and Information Administration: AI Accountability Policy Request for Comment","sanxiyn","2023","blog","LessWrong","www.lesswrong.com/posts/aHxFuJNQ9QNEjmC4f/national-telecommunications-and-information-administration",0,"","governance policy"],["NTIA - AI Accountability Announcement","samshap","2023","blog","LessWrong","www.lesswrong.com/posts/qwq6tkMoeSNKtaej6/ntia-ai-accountability-announcement",0,"","governance"],["Paleontological study of extinctions supports AI as a existential threat to humanity","kpurens","2023","blog","EA Forum","forum.effectivealtruism.org/posts/R5RB4Rjmb2rpHGsmz/paleontological-study-of-extinctions-supports-ai-as-a",0,"",""],["Preliminary investigations on if STEM and EA communities could benefit from more overlap","elteerkers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tgDRDKaMxP9okcrWJ/preliminary-investigations-on-if-stem-and-ea-communities",0,"",""],["Request to AGI organizations: Share your views on pausing AI progress","Akash and simeon_c","2023","blog","LessWrong","www.lesswrong.com/posts/bceeKEnPHSQqgyr36/request-to-agi-organizations-share-your-views-on-pausing-ai",0,"","governance"],["Some Intuitions Around Short AI Timelines Based on Recent Progress","Aaron_Scher","2023","blog","LessWrong","www.lesswrong.com/posts/woZymgKQqB5gEaAAz/some-intuitions-around-short-ai-timelines-based-on-recent",0,"","forecasting"],["Translation: Measures for the Management of Generative Artificial Intelligence Services (Draft for Comment) – April 2023","DigiChina Stanford University","2023","report","digichina.stanford.edu","digichina.stanford.edu/work/translation-measures-for-the-management-of-generative-artificial-intelligence-services-draft-for-comment-april-2023/",0,"",""],["AI Safety Newsletter #1 [CAIS Linkpost]","Akash and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/9ros2kvDGCTuoidqX/ai-safety-newsletter-1-cais-linkpost",0,"",""],["An AI Realist Manifesto: Neither Doomer nor Foomer, but a third more reasonable thing","PashaKamyshev","2023","blog","LessWrong","www.lesswrong.com/posts/jmaZjkzq32pmzoHpF/an-ai-realist-manifesto-neither-doomer-nor-foomer-but-a",0,"","forecasting"],["Current UK government levers on AI development","rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BFBf5yPLoJMGozygE/current-uk-government-levers-on-ai-development",0,"","governance policy"],["Humans are not prepared to operate outside their moral training distribution","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/ZDqiXgSYRgQTSFrsW/humans-are-not-prepared-to-operate-outside-their-moral",0,"","forecasting"],["Misgeneralization as a misnomer","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/04/10/misgeneralization-as-a-misnomer/",0,"",""],["Which stocks or ETFs should you invest in to take advantage of a possible AGI explosion, and why?","BrownHairedEevee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tLuejqs5odxYLvckK/which-stocks-or-etfs-should-you-invest-in-to-take-advantage",0,"",""],["Why I'm not worried about imminent doom","Ariel Kwiatkowski","2023","blog","LessWrong","www.lesswrong.com/posts/Amcb6uokHAmDTRooC/why-i-m-not-worried-about-imminent-doom",0,"","forecasting"],["Why Simulator AIs want to be Active Inference AIs","Jan_Kulveit and rosehadshar","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/YEioD8YLgxih3ydxP/why-simulator-ais-want-to-be-active-inference-ais",0,"","agents theory"],["Agentized LLMs will change the alignment landscape","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/dcoxvEhAfYcov2LA6/agentized-llms-will-change-the-alignment-landscape",0,"","agents"],["Expanding the domain of discourse reveals structure already there but hidden","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mFWBsCy2SHQ97RArb/expanding-the-domain-of-discourse-reveals-structure-already",0,"",""],["Foom seems unlikely in the current LLM training paradigm","Ocracoke","2023","blog","LessWrong","www.lesswrong.com/posts/ktSzxMsKBJmon6FGm/foom-seems-unlikely-in-the-current-llm-training-paradigm",0,"","forecasting"],["Podcast/video/transcript: Eliezer Yudkowsky - Why AI Will Kill Us, Aligning LLMs, Nature of Intelligence, SciFi, & Rationality","PeterSlattery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Wp3EjrgwEBFvnqzvg/podcast-video-transcript-eliezer-yudkowsky-why-ai-will-kill",0,"",""],["All AGI Safety questions welcome (especially basic ones) [April 2023]","steven0461","2023","blog","LessWrong","www.lesswrong.com/posts/wqeStKQ3PGzZaeoje/all-agi-safety-questions-welcome-especially-basic-ones-april-1",0,"",""],["All images from the WaitButWhy sequence on AI","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/kzwAczMyyvnvaAzxq/all-images-from-the-waitbutwhy-sequence-on-ai",0,"","governance"],["Can we evaluate the \"tool versus agent\" AGI prediction?","Ben_West","2023","blog","EA Forum","forum.effectivealtruism.org/posts/a2KEyLaXzBADb8jgg/can-we-evaluate-the-tool-versus-agent-agi-prediction",0,"","evals agents forecasting"],["GPTs are Predictors, not Imitators","Eliezer Yudkowsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nH4c3Q9t9F3nJ7y8W/gpts-are-predictors-not-imitators",0,"",""],["How does a company like Instadeep fit into the current AI landscape?","Tom A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/czfqJ3N5f2AQBZAmL/how-does-a-company-like-instadeep-fit-into-the-current-ai",0,"",""],["Pausing AI Developments Isn't Enough. We Need to Shut it All Down","Eliezer Yudkowsky","2023","blog","LessWrong","www.lesswrong.com/posts/oM9pEezyCb4dCsuKq/pausing-ai-developments-isn-t-enough-we-need-to-shut-it-all-1",0,"","governance"],["Pausing AI Developments Isn’t Enough. We Need to Shut it All Down","Eliezer Yudkowsky","2023","blog","intelligence.org","intelligence.org/2023/04/07/pausing-ai-developments-isnt-enough-we-need-to-shut-it-all-down/",0,"",""],["SERI MATS - Summer 2023 Cohort","Aris and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/aEQBkDPZi6L2LMpnC/seri-mats-summer-2023-cohort",0,"",""],["An 'AGI Emergency Eject Criteria' consensus could be really useful.","tcelferact","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CZvJEhNWjecB8pgzw/an-agi-emergency-eject-criteria-consensus-could-be-really-1",0,"","evals governance"],["Beren's \"Deconfusing Direct vs Amortised Optimisation\"","DragonGod","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5YDczJcLZ6RmN5SSz/beren-s-deconfusing-direct-vs-amortised-optimisation-2",0,"",""],["Environments for Measuring Deception, Resource Acquisition, and Ethical Violations","Dan H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/smDeWfgeYDg9eGq5G/environments-for-measuring-deception-resource-acquisition",0,"","alignment-faking deception power-seeking"],["Goal alignment without alignment on epistemology, ethics, and science is futile","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/fqfAmAGFLKpsnjfJB/goal-alignment-without-alignment-on-epistemology-ethics-and",0,"","agents theory"],["How much should states invest in contingency plans for widespread internet outage?","Kinoshita Yoshikazu (pseudonym)","2023","blog","EA Forum","forum.effectivealtruism.org/posts/knitD2FPQpsTMhLJP/how-much-should-states-invest-in-contingency-plans-for",0,"","policy"],["If Alignment is Hard, then so is Self-Improvement","PavleMiha","2023","blog","LessWrong","www.lesswrong.com/posts/4KHPxsJgGfxwCNSCC/if-alignment-is-hard-then-so-is-self-improvement",0,"",""],["Imagine AGI killed us all in three years. What would have been our biggest mistakes?","anonymous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5hihHtfhGoyNB5kLQ/imagine-agi-killed-us-all-in-three-years-what-would-have",0,"",""],["n=3 AI Risk Quick Math and Reasoning","lionhearted (Sebastian Marshall)","2023","blog","LessWrong","www.lesswrong.com/posts/dcjGrRrXwXBtTBHLn/n-3-ai-risk-quick-math-and-reasoning",0,"","instrumental-convergence forecasting"],["Risks from GPT-4 Byproduct of Recursively Optimizing AIs","ben hayum","2023","blog","LessWrong","www.lesswrong.com/posts/5nfHFRC4RZ6S2zQyb/risks-from-gpt-4-byproduct-of-recursively-optimizing-ais",0,"","instrumental-convergence power-seeking governance"],["Select Agent Specifications as Natural Abstractions","marc/er","2023","blog","LessWrong","www.lesswrong.com/posts/GEYntEDugjawxLTEL/select-agent-specifications-as-natural-abstractions",0,"","agents"],["Should we publish arguments for the preservation of humanity?","Jeremy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GwgnYbyEnAWgBCvwA/should-we-publish-arguments-for-the-preservation-of-humanity",0,"","theory"],["Stampy's AI Safety Info - New Distillations #1 [March 2023] (Expansive interactive FAQ)","markov","2023","blog","LessWrong","www.lesswrong.com/posts/FYRYhkdAQoQibasNB/stampy-s-ai-safety-info-new-distillations-1-march-2023",0,"",""],["Superintelligence Is Not Omniscience","Jeffrey Heninger","2023","blog","aiimpacts.org","aiimpacts.org/superintelligence-is-not-omniscience/",0,"",""],["AISafety.world is a map of the AIS ecosystem","Hamish McDoodles","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mDG49CJyxzeN99ELz/aisafety-world-is-a-map-of-the-ais-ecosystem",0,"",""],["AISafety.world is a map of the AIS ecosystem","Hamish Doodles","2023","blog","LessWrong","www.lesswrong.com/posts/n99LGqGyQWYNyNvXG/aisafety-world-is-a-map-of-the-ais-ecosystem",0,"",""],["Daisy-chaining epsilon-step verifiers","Decaeneus","2023","blog","LessWrong","www.lesswrong.com/posts/nWuvuXeXriyWxrtZd/daisy-chaining-epsilon-step-verifiers",0,"","alignment-faking deception automated-alignment-research"],["Debates on reducing long-term s-risks?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/qJffR9vj92kY32iHg/debates-on-reducing-long-term-s-risks",0,"",""],["Do the Rewards Justify the Means? Measuring Trade-Offs Between Rewards and Ethical Behavior in the Machiavelli Benchmark","Alexander Pan and 9 others","2023","paper","arXiv preprint","arxiv.org/abs/2304.03279",0,"","evals benchmarks deception power-seeking agents"],["I asked my senator to slow AI","Omid","2023","blog","LessWrong","www.lesswrong.com/posts/tcBEh9ZtEWvoHNKZ4/i-asked-my-senator-to-slow-ai",0,"","governance"],["Is \"Recursive Self-Improvement\" Relevant in the Deep Learning Paradigm?","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/oyK6fYYnBi5Nx5pfE/is-recursive-self-improvement-relevant-in-the-deep-learning",0,"","forecasting"],["Is it time for a pause?","Kelsey Piper","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZvMPNLFBHur9qopw9/is-it-time-for-a-pause",0,"","governance"],["Misgeneralization as a misnomer","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dkjwSLfvKwpaQSuWo/misgeneralization-as-a-misnomer",0,"",""],["Some Preliminary Opinions on AI Safety Problems","yonxinzhang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wP2JueKsfvrfNbkT5/some-preliminary-opinions-on-ai-safety-problems",0,"",""],["The Computational Anatomy of Human Values","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pZHpq6dBQzCZjjMgM/the-computational-anatomy-of-human-values",0,"",""],["Where to begin in ML/AI?","Jake the Student","2023","blog","LessWrong","www.lesswrong.com/posts/YPZgWeRR3W5zgsyL7/where-to-begin-in-ml-ai",0,"",""],["Yoshua Bengio: \"Slowing down development of AI systems passing the Turing test\"","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/tq8uMdSDj8iRnGmTE/yoshua-bengio-slowing-down-development-of-ai-systems-passing",0,"","governance"],["\"Corrigibility at some small length\" by dath ilan","Christopher King","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5sRK4rXH2EeSQJCau/corrigibility-at-some-small-length-by-dath-ilan",0,"",""],["Best arguments against instrumental convergence?","lfrymire","2023","blog","LessWrong","www.lesswrong.com/posts/zDmbtt7o4J8nY3d7L/best-arguments-against-instrumental-convergence",0,"","instrumental-convergence"],["Empathy bandaid for immediate AI catastrophe","installgentoo","2023","blog","LessWrong","www.lesswrong.com/posts/xRSSM3qJu7HDEbKnG/empathy-bandaid-for-immediate-ai-catastrophe",0,"",""],["OpenAI: Our approach to AI safety","g-w1","2023","blog","LessWrong","www.lesswrong.com/posts/ceiR9bbupYiyBq4hM/openai-our-approach-to-ai-safety",0,"",""],["The Orthogonality Thesis is Not Obviously True","Omnizoid","2023","blog","EA Forum","forum.effectivealtruism.org/posts/e2dK25iWou3irqFss/the-orthogonality-thesis-is-not-obviously-true",0,"",""],["Universality and Hidden Information in Concept Bottleneck Models","Hoagy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uTPetRBbkP4p6FhGf/universality-and-hidden-information-in-concept-bottleneck",0,"","interpretability"],["What to suggest companies & entrepreneurs do to use AI safely?","AlfalfaBloom","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DE6bafBMfiLmXEapa/what-to-suggest-companies-and-entrepreneurs-do-to-use-ai",0,"","governance"],["AI Summer Harvest","Cleo Nardo","2023","blog","LessWrong","www.lesswrong.com/posts/P98i7kAN2uWuy7mhD/ai-summer-harvest",0,"","governance"],["Excessive AI growth-rate yields little socio-economic benefit.","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WbdLYgbpxfrSXCBS6/excessive-ai-growth-rate-yields-little-socio-economic",0,"","governance forecasting"],["Giant (In)scrutable Matrices: (Maybe) the Best of All Possible Worlds","1a3orn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GkC6YTu4DWp2zwf9k/giant-in-scrutable-matrices-maybe-the-best-of-all-possible",0,"","interpretability"],["Keep Chasing AI Safety Press Coverage","RedStateBlueState","2023","blog","EA Forum","forum.effectivealtruism.org/posts/u3cJGX33zf32TsCMg/keep-chasing-ai-safety-press-coverage",0,"",""],["Penalize Model Complexity Via Self-Distillation","research_prime_space","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fzGbKHbSytXH5SKTN/penalize-model-complexity-via-self-distillation",0,"",""],["Why might AI be a x-risk? Succinct explanations please","Sanjay","2023","blog","EA Forum","forum.effectivealtruism.org/posts/exQZAgfXSxXECki5T/why-might-ai-be-a-x-risk-succinct-explanations-please",0,"",""],["AI Control idea: Give an AGI the primary objective of deleting itself, but construct obstacles to this as best we can. All other objectives are secondary to this primary goal.","Justausername","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2xjrJwbmsaGzjD7w7/ai-control-idea-give-an-agi-the-primary-objective-of",0,"","ai-control"],["Apply to the Cavendish Labs Fellowship (by 4/15)","Derik K and dyusha","2023","blog","EA Forum","forum.effectivealtruism.org/posts/S8xfeJGER74xwqLta/apply-to-the-cavendish-labs-fellowship-by-4-15",0,"",""],["Exploratory Analysis of RLHF Transformers with TransformerLens","Curt Tigges","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ky3WnDwQbLAucGrXf/exploratory-analysis-of-rlhf-transformers-with",0,"","rlhf interpretability"],["If interpretability research goes well, it may get dangerous","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BinkknLBYxskMXuME/if-interpretability-research-goes-well-it-may-get-dangerous",0,"","interpretability"],["Import AI 323: AI researcher warns about AI; BloombergGPT; and an open source Flamingo","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-323-ai-researcher-warns",0,"",""],["Mati's introduction to pausing giant AI experiments","Mati_Roy","2023","blog","LessWrong","www.lesswrong.com/posts/jhSKmmhyjEyBrjbix/mati-s-introduction-to-pausing-giant-ai-experiments",0,"",""],["Platform for Project Spitballing? (e.g., for AI field building)","Harrison Durland","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Zcy8EDfQ9TXFGL75m/platform-for-project-spitballing-e-g-for-ai-field-building",0,"",""],["Reducing profit motivations in AI development","Luke Frymire","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gvN4LBh7ZMguxxhHW/reducing-profit-motivations-in-ai-development",0,"","governance policy"],["Repeated Play of Imperfect Newcomb's Paradox in Infra-Bayesian Physicalism","Sven Nilsen","2023","blog","LessWrong","www.lesswrong.com/posts/QbXgTXtHGEZyB8Wjn/repeated-play-of-imperfect-newcomb-s-paradox-in-infra",0,"","agents theory"],["Towards empathy in RL agents and beyond: Insights from cognitive science for AI Alignment","Marc Carauleanu","2023","blog","LessWrong","www.lesswrong.com/posts/bNpqBNvfgCWixB2MT/towards-empathy-in-rl-agents-and-beyond-insights-from-1",0,"","agents"],["Advanced AI can beat humanity","Loppukilpailija","2023","blog","LessWrong","www.lesswrong.com/posts/PoB9iWS45vsYRa7Ty/advanced-ai-can-beat-humanity",0,"",""],["AISC 2023, Progress Report for March: Team Interpretable Architectures","Robert Kralisch and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/6JPZcScxLZtYzYRrQ/aisc-2023-progress-report-for-march-team-interpretable",0,"","interpretability"],["Exploratory Analysis of TRLX RLHF Transformers with TransformerLens","Curt Tigges","2023","blog","blog.eleuther.ai","blog.eleuther.ai/trlx-exploratory-analysis/",0,"","rlhf"],["Pessimism about AI Safety","Max_He-Ho","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uiBCfZH7NrujeLdgK/pessimism-about-ai-safety",0,"","governance"],["Pessimism about AI Safety","Max_He-Ho and Peter Kuhn","2023","blog","LessWrong","www.lesswrong.com/posts/vm97ZtpwRbGrjNwha/pessimism-about-ai-safety",0,"","governance"],["Predictions for future AI governance?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/efmn6fydPcymxNZpT/predictions-for-future-ai-governance",0,"","governance forecasting"],["Research Summary: Forecasting with Large Language Models","Damien Laird","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AW4iRhriRHkdGokLp/research-summary-forecasting-with-large-language-models",0,"","forecasting"],["Transparency for Generalizing Alignment from Toy Models","Johannes C. Mayer","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9iHwqnH4ZeqkGDbrb/transparency-for-generalizing-alignment-from-toy-models-1",0,"","interpretability"],["Ultimate ends may be easily hidable behind convergent subgoals","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8qCKZj24FJotm3EKd/ultimate-ends-may-be-easily-hidable-behind-convergent",0,"",""],["[New LW Feature] \"Debates\"","Ruby and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/kXiAGRWFquXFMi68Y/new-lw-feature-debates",0,"",""],["A policy guaranteed to increase AI timelines","richardkorzekwa","2023","blog","aiimpacts.org","aiimpacts.org/a-policy-guaranteed-to-increase-ai-timelines/",0,"","policy forecasting"],["AI community building: EliezerKart","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/9SBSTFECnyHpkKyBA/ai-community-building-eliezerkart",0,"","governance"],["AI Safety via Luck","Jozdien","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rH9sXupnoR8wSmRe9/ai-safety-via-luck-2",0,"",""],["Campaign for AI Safety: Please join me","Nik Samoylov","2023","blog","LessWrong","www.lesswrong.com/posts/SZ3NcxYYNp5fnE74a/campaign-for-ai-safety-please-join-me",0,"",""],["How to persuade a non-CS background person to believe AGI is 50% possible in 2040?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oNeX5x362cWZX84PT/how-to-persuade-a-non-cs-background-person-to-believe-agi-is",0,"","forecasting"],["Pillars to Convergence","Phlobton","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CvrWR7LSw2GpJMsPp/pillars-to-convergence",0,"",""],["Policy discussions follow strong contextualizing norms","Richard_Ngo","2023","blog","LessWrong","www.lesswrong.com/posts/cbQih72wbKkrSX7yx/policy-discussions-follow-strong-contextualizing-norms",0,"","governance policy"],["Singularities against the Singularity: Announcing Workshop on Singular Learning Theory and Alignment","Jesse Hoogland and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HtxLbGvD7htCybLmZ/singularities-against-the-singularity-announcing-workshop-on",0,"",""],["Vael Gates: Risks from Highly-Capable AI (March 2023)","Vael Gates","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WqQDKKgZTdFe6GAFq/vael-gates-risks-from-highly-capable-ai-march-2023-1",0,"",""],["Γαμινγκ the Algorithms: Large Language Models as Mirrors","Haris Shekeris","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KoriEzLMz5kxbHLEm/gamingk-the-algorithms-large-language-models-as-mirrors",0,"",""],["AI, Cybersecurity, and Malware: A Shallow Report [General]","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TxeqEJSmNdBKq9Ekw/ai-cybersecurity-and-malware-a-shallow-report-general",0,"",""],["AI, Cybersecurity, and Malware: A Shallow Report [Technical]","Madhav Malhotra","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SMsobbG7tgya2neN9/ai-cybersecurity-and-malware-a-shallow-report-technical",0,"",""],["ChatGPT banned in Italy over privacy concerns","Ollie J","2023","blog","LessWrong","www.lesswrong.com/posts/FAwAqtzp53vjFAd4k/chatgpt-banned-in-italy-over-privacy-concerns",0,"","governance"],["Critiques of prominent AI safety labs: Redwood Research","Omega","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DaRvpDHHdaoad9Tfu/critiques-of-prominent-ai-safety-labs-redwood-research",0,"","robustness"],["GPT-4 busted? Clear self-interest when summarizing articles about itself vs when article talks about Claude, LLaMA, or DALL·E 2","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/n8gHxkwHuErfjMDdc/gpt-4-busted-clear-self-interest-when-summarizing-articles",0,"","rlhf"],["Human Values and AGI Risk | William James","William James","2023","blog","EA Forum","forum.effectivealtruism.org/posts/YPonS6QpDwhbRnT8N/human-values-and-agi-risk-or-william-james",0,"",""],["Imagine a world where Microsoft employees used Bing","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/nsGgYL2umBrqW7Jbj/imagine-a-world-where-microsoft-employees-used-bing",0,"",""],["It Can't Be Mesa-Optimizers All The Way Down (Or Else It Can't Be Long-Term Supercoherence?)","Austin Witte","2023","blog","LessWrong","www.lesswrong.com/posts/aAtpQjEaX3Fwzcfdj/it-can-t-be-mesa-optimizers-all-the-way-down-or-else-it-can",0,"",""],["Keep Making AI Safety News","RedStateBlueState","2023","blog","EA Forum","forum.effectivealtruism.org/posts/K6ugqr4cWnB6KELh4/keep-making-ai-safety-news",0,"",""],["Manifund x AI Worldviews","Austin and Rachel Weinberg","2023","blog","EA Forum","forum.effectivealtruism.org/posts/o3zdvskr2DZPTRnkF/manifund-x-ai-worldviews",0,"",""],["Maze-solving agents: Add a top-right vector, make the agent go to the top-right","TurnTrout and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gRp6FAWcQiCWkouN5/maze-solving-agents-add-a-top-right-vector-make-the-agent-go",0,"","interpretability agents"],["Seeking advice on impactful career paths given my unique capabilities and interests","Grateful4PathTips (bot)","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CgffHE7tggZmw5Lb6/seeking-advice-on-impactful-career-paths-given-my-unique",0,"",""],["What are the biggest obstacles on AI safety research career?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CMsdrbq9zQtHwvcdE/what-are-the-biggest-obstacles-on-ai-safety-research-career",0,"",""],["Widening Overton Window - Open Thread","Prometheus","2023","blog","LessWrong","www.lesswrong.com/posts/S3bpZvLWta7wbkJi4/widening-overton-window-open-thread",0,"","governance"],["[Event] Join Metaculus Tomorrow, March 31st, for Forecast Friday!","christian","2023","blog","EA Forum","forum.effectivealtruism.org/posts/axJcQL6ownMQQD5AA/event-join-metaculus-tomorrow-march-31st-for-forecast-friday",0,"","forecasting"],["AI and Evolution","Dan H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TxHBeEMC7SBZvxCk8/ai-and-evolution",0,"",""],["ChatGPT is capable of cognitive empathy!","mikbp","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2qZSv3skTT3pgEcGu/chatgpt-is-capable-of-cognitive-empathy",0,"",""],["Deference on AI timelines: survey results","Sam Clarke and mccaffary","2023","blog","LessWrong","www.lesswrong.com/posts/qccxb3uzwFDsRuJuP/deference-on-ai-timelines-survey-results",0,"","forecasting"],["How is AI governed and regulated, around the world?","Mitchell_Porter","2023","blog","LessWrong","www.lesswrong.com/posts/ogC8agX64nW6N9EEW/how-is-ai-governed-and-regulated-around-the-world",0,"","governance"],["Imitation Learning from Language Feedback","Jérémy Scheurer and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mCZSXdZoNoWn5SkvE/imitation-learning-from-language-feedback-1",0,"","rlhf chain-of-thought-faithfulness"],["Longtermism and shorttermism can disagree on nuclear war to stop advanced AI","David Johnston","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WSCAck83BbYR5ZyRF/longtermism-and-shorttermism-can-disagree-on-nuclear-war-to",0,"",""],["Nuclear brinksmanship is not a good AI x-risk strategy","titotal","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Rn4Em42vXcDWCEhSK/nuclear-brinksmanship-is-not-a-good-ai-x-risk-strategy",0,"","governance robustness"],["Recruit the World’s best for AGI Alignment","Greg_Colbourn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CpyAgKt4gRza7npYf/recruit-the-world-s-best-for-agi-alignment",0,"",""],["Role Architectures: Applying LLMs to consequential tasks","Eric Drexler","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/AKaf8zN2neXQEvLit/role-architectures-applying-llms-to-consequential-tasks",0,"","deception"],["The 0.2 OOMs/year target","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9xfRjaKDTb57BaGWv/the-0-2-ooms-year-target",0,"","governance forecasting"],["You Can’t Predict a Game of Pinball","Jeffrey Heninger","2023","blog","aiimpacts.org","aiimpacts.org/you-cant-predict-a-game-of-pinball/",0,"",""],["\"Sorcerer's Apprentice\" from Fantasia as an analogy for alignment","awg","2023","blog","LessWrong","www.lesswrong.com/posts/FzBZijmitZuasJwoq/sorcerer-s-apprentice-from-fantasia-as-an-analogy-for",0,"",""],["Actually, Othello-GPT Has A Linear Emergent World Representation","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nmxzr2zsjNtjaHh7x/actually-othello-gpt-has-a-linear-emergent-world",0,"","interpretability"],["Desensitizing Deepfakes","Phib","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bfDke8yv6sX94jF4R/desensitizing-deepfakes",0,"",""],["Nobody’s on the ball on AGI alignment","leopold","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5LNxeWFdoynvgZeik/nobody-s-on-the-ball-on-agi-alignment",0,"","evals governance"],["Othello-GPT: Future Work I Am Excited About","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qgK7smTvJ4DB8rZ6h/othello-gpt-future-work-i-am-excited-about",0,"","interpretability"],["Othello-GPT: Reflections on the Research Process","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TAz44Lb9n9yf52pv8/othello-gpt-reflections-on-the-research-process",0,"","interpretability"],["Pausing AI Developments Isn't Enough. We Need to Shut it All Down by Eliezer Yudkowsky","jacquesthibs","2023","blog","LessWrong","www.lesswrong.com/posts/Aq5X9tapacnk2QGY4/pausing-ai-developments-isn-t-enough-we-need-to-shut-it-all",0,"","governance"],["Strong Cheap Signals","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/m8fHNfrhfZuHpNJhk/strong-cheap-signals",0,"","theory"],["Want to win the AGI race? Solve alignment.","leopold","2023","blog","LessWrong","www.lesswrong.com/posts/zDf7fnentCFTdK3K6/want-to-win-the-agi-race-solve-alignment",0,"","governance"],["What are the arguments that support China building AGI+ if Western companies delay/pause AI development?","DMMF","2023","blog","EA Forum","forum.effectivealtruism.org/posts/w5GsJBF8YHqWdCroW/what-are-the-arguments-that-support-china-building-agi-if",0,"","governance policy"],["“Unintentional AI safety research”: Why not systematically mine AI technical research for safety purposes?","ghostwheel","2023","blog","LessWrong","www.lesswrong.com/posts/GBNayXzcboJumL2Dx/unintentional-ai-safety-research-why-not-systematically-mine",0,"","automated-alignment-research"],["100 Dinners And A Workshop: Information Preservation And Goals","Stephen Fowler","2023","blog","LessWrong","www.lesswrong.com/posts/Jj2cThYbNfgZPQkMw/100-dinners-and-a-workshop-information-preservation-and",0,"","agents theory"],["A rough and incomplete review of some of John Wentworth's research","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mgjHS6ou7DgwhKPpu/a-rough-and-incomplete-review-of-some-of-john-wentworth-s",0,"",""],["Corrigibility, Self-Deletion, and Identical Strawberries","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/gcK47FGBAzgc4oDSr/corrigibility-self-deletion-and-identical-strawberries",0,"",""],["Explorers in a virtual country: Navigating the knowledge landscape of large language models","AlexanderSaeri","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vYdgmvEQaMcstXcNv/explorers-in-a-virtual-country-navigating-the-knowledge",0,"",""],["Governing High-Impact AI Systems: Understanding Canada’s Proposed AI Bill. April 15, Carleton University, Ottawa","Liav Koren","2023","blog","LessWrong","www.lesswrong.com/posts/a3fvua59H6nktCtxw/governing-high-impact-ai-systems-understanding-canada-s",0,"","governance"],["I had a chat with GPT-4 on the future of AI and AI safety","Kristian Freed","2023","blog","LessWrong","www.lesswrong.com/posts/WCJWwgrNnTDvf2aJr/i-had-a-chat-with-gpt-4-on-the-future-of-ai-and-ai-safety",0,"",""],["Improving Code Generation by Training with Natural Language Feedback","Angelica Chen and 7 others","2023","paper","arXiv preprint","arxiv.org/abs/2303.16749",0,"","benchmarks"],["It Looks Like You’re Trying To Take Over The World","Gwern Branwen","2023","blog","gwern.net","www.gwern.net/Clippy.page",0,"",""],["Natural Selection Favors AIs over Humans","Dan Hendrycks","2023","paper","arXiv preprint","arxiv.org/abs/2303.16200",0,"","agents"],["Some of My Current Impressions Entering AI Safety","Phib","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iKbdvrdFRSsPwNkh7/some-of-my-current-impressions-entering-ai-safety",0,"","forecasting"],["Training Language Models with Language Feedback at Scale","Jérémy Scheurer and 6 others","2023","paper","arXiv preprint","arxiv.org/abs/2303.16755",0,"","rlhf evals robustness"],["What longtermist projects would you like to see implemented?","Buhl","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TYyHpiAQ3TetwRMHC/what-longtermist-projects-would-you-like-to-see-implemented",0,"","policy"],["When Will We Spend Enough to Train Transformative AI","Skye Nygaard","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bL3riEPKqZKjdHmFg/when-will-we-spend-enough-to-train-transformative-ai",0,"","forecasting"],["Why does advanced AI want not to be shut down?","RedFishBlueFish","2023","blog","LessWrong","www.lesswrong.com/posts/aohCk8JeDGiuvGiFZ/why-does-advanced-ai-want-not-to-be-shut-down",0,"",""],["Are there cause priortizations estimates for s-risks supporters?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rE7t2qo72uoghEhna/are-there-cause-priortizations-estimates-for-s-risks",0,"",""],["Best resources to learn philosophy of mind and AI?","Sky Moo","2023","blog","LessWrong","www.lesswrong.com/posts/7LQYAWiJsp2MSygXG/best-resources-to-learn-philosophy-of-mind-and-ai",0,"",""],["CAIS-inspired approach towards safer and more interpretable AGIs","Peter Hroššo","2023","blog","LessWrong","www.lesswrong.com/posts/6DSL5WbTFiz4zcf8u/cais-inspired-approach-towards-safer-and-more-interpretable",0,"","interpretability governance chain-of-thought-faithfulness"],["ChatGPT bug leaked users' conversation histories","Ian Turner","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KBtyxTZCMzh2BJnJs/chatgpt-bug-leaked-users-conversation-histories",0,"","governance"],["Governing High-Impact AI Systems: Understanding Canada’s Proposed AI Bill. April 15, Carleton University, Ottawa","Liav.Koren","2023","blog","EA Forum","forum.effectivealtruism.org/posts/J7LwudixQX5FHrcP3/governing-high-impact-ai-systems-understanding-canada-s-2",0,"","governance policy"],["GPT-4 is bad at strategic thinking","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/ekBsq6mATCvEvqmkC/gpt-4-is-bad-at-strategic-thinking",0,"",""],["Import AI 322: Huawei's trillion parameter model; AI systems as moral patients; parasocial bots via Character.ai","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-322-huaweis-trillion-parameter",0,"",""],["Lessons from Convergent Evolution for AI Alignment","Jan_Kulveit and rosehadshar","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sam4ehxHgnJEGCKed/lessons-from-convergent-evolution-for-ai-alignment",0,"","instrumental-convergence"],["New blog: Planned Obsolescence","Ajeya and Kelsey Piper","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Si3H2oM8tYwcnBWdS/new-blog-planned-obsolescence",0,"",""],["Nobody knows how to reliably test for AI safety","marcusarvan","2023","blog","LessWrong","www.lesswrong.com/posts/jbTxWBYCLHm7exnY7/nobody-knows-how-to-reliably-test-for-ai-safety",0,"",""],["Please help me sense-check my assumptions about the needs of the AI Safety community and related career plans","PeterSlattery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5HhHJjP5qkCwvJozA/please-help-me-sense-check-my-assumptions-about-the-needs-of",0,"",""],["Practical Pitfalls of Causal Scrubbing","Jérémy Scheurer and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DFarDnQjMnjsKvW8s/practical-pitfalls-of-causal-scrubbing",0,"","interpretability robustness"],["The Prospect of an AI Winter","Erich_Grunewald","2023","blog","LessWrong","www.lesswrong.com/posts/6xRsdig9FXfGJdinX/the-prospect-of-an-ai-winter",0,"","forecasting"],["Descriptive vs. specifiable values","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tTLhEqaWJGwjLaop3/descriptive-vs-specifiable-values",0,"",""],["EleutherAI Second Retrospective: The long version","Stella Biderman and 5 others","2023","blog","blog.eleuther.ai","blog.eleuther.ai/year-two-full/",0,"",""],["LLM Modularity: The Separability of Capabilities in Large Language Models","NickyP","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/j84JhErNezMxyK4dH/llm-modularity-the-separability-of-capabilities-in-large",0,"","interpretability"],["The alignment stability problem","Seth Herd","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/g3pbJPQpNJyFfbHKd/the-alignment-stability-problem",0,"","agents"],["What happens with logical induction when...","Donald Hobson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iqRmz5JipwMXg4Trc/what-happens-with-logical-induction-when",0,"","theory"],["What would a compute monitoring plan look like? [Linkpost]","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/ByCwWRgvTsSC6Wxst/what-would-a-compute-monitoring-plan-look-like-linkpost",0,"","governance monitoring"],["$500 Bounty/Contest: Explain Infra-Bayes In The Language Of Game Theory","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ig62gqK2xRPv79Wsr/usd500-bounty-contest-explain-infra-bayes-in-the-language-of",0,"",""],["A stylized dialogue on John Wentworth's claims about markets and optimization","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fJBTRa7m7KnCDdzG5/a-stylized-dialogue-on-john-wentworth-s-claims-about-markets",0,"",""],["Aligned AI as a wrapper around an LLM","cousin_it","2023","blog","LessWrong","www.lesswrong.com/posts/3KmuJavii9njiDtGZ/aligned-ai-as-a-wrapper-around-an-llm",0,"",""],["Can independent researchers get a sponsored visa for the US or UK?","jacquesthibs","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gEofudCuTTESGPypc/can-independent-researchers-get-a-sponsored-visa-for-the-us",0,"",""],["Good News, Everyone!","jbash","2023","blog","LessWrong","www.lesswrong.com/posts/CPKYuJqLYGpBTtdFd/good-news-everyone",0,"","robustness"],["Mitigating existential risks associated with human nature and AI: Thoughts on serious measures.","Paul J. Watson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NkAPQnDuSMDLwziYY/mitigating-existential-risks-associated-with-human-nature",0,"",""],["My attempt at explaining the case for AI risk in a straightforward way","JulianHazell","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FdYzaKxaP7hJNa5AF/my-attempt-at-explaining-the-case-for-ai-risk-in-a",0,"",""],["opinions on the consequences of AI","Tamsin Leake","2023","blog","carado.moe","carado.moe/map-opinions-ai.html",0,"",""],["Are extrapolation-based AIs alignable?","cousin_it","2023","blog","LessWrong","www.lesswrong.com/posts/xQKHgEq9YrvzKiABA/are-extrapolation-based-ais-alignable",0,"",""],["Does GPT-4 exhibit agency when summarizing articles?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/dzTmFLC93PRzJEDcW/does-gpt-4-exhibit-agency-when-summarizing-articles",0,"",""],["Exploring Metaculus’ community predictions","Vasco Grilo","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zeL52MFB2Pkq9Kdme/exploring-metaculus-community-predictions",0,"","forecasting"],["GPT-2005: A conversation with ChatGPT (featuring semi-functional Wolfram Alpha plugin!)","Lone Pine","2023","blog","LessWrong","www.lesswrong.com/posts/bfsyLY3Xnq442eKL8/gpt-2005-a-conversation-with-chatgpt-featuring-semi",0,"","forecasting"],["Grinding slimes in the dungeon of AI alignment research","Max H","2023","blog","LessWrong","www.lesswrong.com/posts/DcyGj7yn52qS5DPna/grinding-slimes-in-the-dungeon-of-ai-alignment-research",0,"",""],["Metaculus Predicts Weak AGI in 2 Years and AGI in 10","Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AtdApEsvPr8QhdoBa/metaculus-predicts-weak-agi-in-2-years-and-agi-in-10",0,"","forecasting"],["More experiments in GPT-4 agency: writing memos","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/vRjYokhT22EvLHZCJ/more-experiments-in-gpt-4-agency-writing-memos",0,"",""],["The Concept of Boundary Layer in Language Games and Its Implications for AI","Mirage","2023","blog","EA Forum","forum.effectivealtruism.org/posts/743io3WyzezJbHowW/the-concept-of-boundary-layer-in-language-games-and-its",0,"",""],["Wittgenstein and ML — parameters vs architecture","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tiKG7gvQ33vf8QAgy/wittgenstein-and-ml-parameters-vs-architecture",0,"","interpretability"],["continue working on hard alignment! don't give up!","Tamsin Leake","2023","blog","carado.moe","carado.moe/continue-working-hard-alignment.html",0,"",""],["EAI Alignment Speaker Series #1: Challenges for Safe & Beneficial Brain-Like Artificial General Intelligence with Steve Byrnes","Curtis Huebner and Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ajhtyKxtmmErTwH5t/eai-alignment-speaker-series-1-challenges-for-safe-and",0,"",""],["GPT-4 aligning with acasual decision theory when instructed to play games, but includes a CDT explanation that's incorrect if they differ","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/hsbAHvRzxTpLTnb2D/gpt-4-aligning-with-acasual-decision-theory-when-instructed",0,"","alignment-faking deception theory"],["Join the AI governance and interpretability hackathons!","Esben Kran and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/JQnYZghxrdpEYHaB8/join-the-ai-governance-and-interpretability-hackathons",0,"","interpretability governance"],["The Overton Window widens: Examples of AI risk in the media","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/SvwuduvpsKtXkLnPF/the-overton-window-widens-examples-of-ai-risk-in-the-media",0,"",""],["The Quantization Model of Neural Scaling","Eric J. Michaud and 3 others","2023","paper","arXiv preprint","arxiv.org/abs/2303.13506",0,"","scaling-laws"],["Truth and Advantage: Response to a draft of “AI safety seems hard to measure”","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/03/22/truth-and-advantage-response-to-a-draft-of-ai-safety-seems-hard-to-measure/",0,"",""],["[Linkpost] Shorter version of report on existential risk from power-seeking AI","Joe_Carlsmith","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BzyeByJcdBqiyjGaG/linkpost-shorter-version-of-report-on-existential-risk-from",0,"","power-seeking"],["[Linkpost] Shorter version of report on existential risk from power-seeking AI","Joe Carlsmith","2023","blog","LessWrong","www.lesswrong.com/posts/cERDWc78ZQdz7QYKD/linkpost-shorter-version-of-report-on-existential-risk-from",0,"","power-seeking"],["Announcing the European Network for AI Safety (ENAIS)","Esben Kran and 5 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/92TAmcppCL7t54Ajn/announcing-the-european-network-for-ai-safety-enais",0,"",""],["Empirical risk minimization is fundamentally confused","Jesse Hoogland","2023","blog","LessWrong","www.lesswrong.com/posts/zuYRyC3zghzgXLpEW/empirical-risk-minimization-is-fundamentally-confused",0,"","interpretability"],["Is Bill Gates overly optomistic about AI?","Dov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xSHWmarG4AbxKmCrF/is-bill-gates-overly-optomistic-about-ai",0,"",""],["Key Questions for Digital Minds","Jacy Reese Anthis","2023","blog","LessWrong","www.lesswrong.com/posts/S3EgMfDGkrA8WCvep/key-questions-for-digital-minds-3",0,"","forecasting"],["Part 3: A Proposed Approach for AI Safety Movement Building: Projects, Professions, Skills, and Ideas for the Future [long post][bounty for feedback]","PeterSlattery","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8XZmu8BM5JBtSnHiP/part-3-a-proposed-approach-for-ai-safety-movement-building",0,"",""],["The space of systems and the space of maps","Jan_Kulveit and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/b9sGz74ayftqPBDYv/the-space-of-systems-and-the-space-of-maps",0,"",""],["The space of systems and the space of maps","Jan_Kulveit and 3 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/d4W4inHhts5Y2szBf/the-space-of-systems-and-the-space-of-maps",0,"",""],["Truth and Advantage: Response to a draft of \"AI safety seems hard to measure\"","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/SpDHvbcJsiE5mxBzj/truth-and-advantage-response-to-a-draft-of-ai-safety-seems",0,"",""],["Whether you should do a PhD doesn't depend much on timelines.","alex lawsen (previously alexrjl)","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jfLjsxcejCFDpo7dw/whether-you-should-do-a-phd-doesn-t-depend-much-on-timelines",0,"","forecasting"],["Why AI Safety is Hard","Simon Möller","2023","blog","LessWrong","www.lesswrong.com/posts/7pCBPPFYgG7nBiNbL/why-ai-safety-is-hard",0,"",""],["\"Aligned with who?\" Results of surveying 1,000 US participants on AI values","Holly Morgan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DpQc5LzwdbZrMHzRa/aligned-with-who-results-of-surveying-1-000-us-participants",0,"","governance"],["Capabilities Denial: The Danger of Underestimating AI","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/vjfgb8Woy7vF8KdT5/capabilities-denial-the-danger-of-underestimating-ai",0,"",""],["Capping AGI profits","Luke Frymire","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hXLGQBrDg3iwQjWDp/capping-agi-profits",0,"","governance policy"],["Clarifying mesa-optimization","Marius Hobbhahn and Pierre Peigné","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NpJkFLBJEq7JQt7oy/clarifying-mesa-optimization",0,"",""],["Deep Deceptiveness","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/XWwvwytieLtEWaFJX/deep-deceptiveness",0,"","alignment-faking deception"],["Deep Deceptiveness","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/03/21/deep-deceptiveness/",0,"","deception"],["Future Matters #8: Bing Chat, AI labs on safety, and pausing Future Matters","Pablo and matthew.vandermerwe","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CrmE6T5A8JhkxnRzw/future-matters-8-bing-chat-ai-labs-on-safety-and-pausing",0,"","forecasting"],["My Objections to \"We’re All Gonna Die with Eliezer Yudkowsky\"","Quintin Pope","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wAczufCpMdaamF9fy/my-objections-to-we-re-all-gonna-die-with-eliezer-yudkowsky",0,"",""],["New 'South Park' episode on AI & Chat GPT","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dfwuHD69keSEj6mBN/new-south-park-episode-on-ai-and-chat-gpt",0,"",""],["Some constructions for proof-based cooperation without Löb","James Payor","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/SBahPHStddcFJnyft/some-constructions-for-proof-based-cooperation-without-loeb",0,"",""],["the QACI alignment plan: table of contents","Tamsin Leake","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4RrLiboiGGKfsanMF/the-qaci-alignment-plan-table-of-contents",0,"",""],["Where I'm at with AI risk: convinced of danger but not (yet) of doom","Amber Dawn","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fXkCcsyF8M6dp6sXx/where-i-m-at-with-ai-risk-convinced-of-danger-but-not-yet-of",0,"","instrumental-convergence"],["Import AI 321: Open source GPT3; giving away democracy to AGI companies; GPT-4 is a political artifact","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-321-open-source-gpt3-giving",0,"",""],["RLHF does not appear to differentially cause mode-collapse","Arthur Conmy and beren","2023","blog","LessWrong","www.lesswrong.com/posts/pjesEx526ngE6dnmr/rlhf-does-not-appear-to-differentially-cause-mode-collapse",0,"","rlhf"],["Sentience in Machines - How Do We Test for This Objectively?","Mayowa Osibodu","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HtiyM6KQFogL77oBJ/sentience-in-machines-how-do-we-test-for-this-objectively",0,"","interpretability"],["The Natural State is Goodhart","devansh","2023","blog","LessWrong","www.lesswrong.com/posts/gB6rXMy63LNYkycrt/the-natural-state-is-goodhart",0,"","goodharts-law"],["the QACI alignment plan: table of contents","Tamsin Leake","2023","blog","carado.moe","carado.moe/qaci.html",0,"",""],["The Wizard of Oz Problem: How incentives and narratives can skew our perception of AI developments","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/7LLLkMGq4ncinzrmd/the-wizard-of-oz-problem-how-incentives-and-narratives-can",0,"","governance"],["\"Wide\" vs \"Tall\" superintelligence","Templarrr","2023","blog","LessWrong","www.lesswrong.com/posts/tggt6cWtBFCYJrbw8/wide-vs-tall-superintelligence",0,"",""],["A tension between two prosaic alignment subgoals","Alex Lawsen","2023","blog","LessWrong","www.lesswrong.com/posts/Wvtri2ooQyFC6sxPB/a-tension-between-two-prosaic-alignment-subgoals-1",0,"","alignment-faking deception"],["How AI could workaround goals if rated by people","ProgramCrafter","2023","blog","LessWrong","www.lesswrong.com/posts/8vf3wKt8d4qP4CCQk/how-ai-could-workaround-goals-if-rated-by-people",0,"","alignment-faking deception"],["How much should governments pay to prevent catastrophes? Longtermism’s limited role","EJT and CarlShulman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DiGL5FuLgWActPBsf/how-much-should-governments-pay-to-prevent-catastrophes",0,"","policy"],["More information about the dangerous capability evaluations we did with GPT-4 and Claude.","Beth Barnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4Gt42jX7RiaNaxCwP/more-information-about-the-dangerous-capability-evaluations",0,"","evals"],["Probabilistic Payor Lemma?","abramdemski","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZWhJcHPmRaXAPAK5k/probabilistic-payor-lemma",0,"",""],["QACI blob location: an issue with firstness","Tamsin Leake","2023","blog","carado.moe","carado.moe/blob-quantum-issue.html",0,"",""],["Shell games","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zqmAMst8hmsdJqrpR/shell-games",0,"",""],["you can't simulate the universe from the beginning?","Tamsin Leake","2023","blog","carado.moe","carado.moe/cant-simulate-the-universe.html",0,"",""],["\"Publish or Perish\" (a quick note on why you should try to make your work legible to existing academic communities)","David Scott Krueger (formerly: capybaralet)","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LKAogXdruuZXdx6ZH/publish-or-perish-a-quick-note-on-why-you-should-try-to-make",0,"",""],["An Appeal to AI Superintelligence: Reasons to Preserve Humanity","James_Miller","2023","blog","LessWrong","www.lesswrong.com/posts/azRwPDbZfpadoL7WW/an-appeal-to-ai-superintelligence-reasons-to-preserve",0,"",""],["Potential employees have a unique lever to influence the behaviors of AI labs","oxalis","2023","blog","EA Forum","forum.effectivealtruism.org/posts/PMFoxr62AeLEwPAH9/potential-employees-have-a-unique-lever-to-influence-the",0,"",""],["Pros and Cons of boycotting paid Chat GPT","NickLaing","2023","blog","EA Forum","forum.effectivealtruism.org/posts/khRpZcCzjHPj2qLL5/pros-and-cons-of-boycotting-paid-chat-gpt",0,"",""],["Would you pursue software engineering as a career today?","justaperson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FgHa7FyiPyvBmMvS7/would-you-pursue-software-engineering-as-a-career-today",0,"",""],["[Linkpost] Alpaca 7B release | Budget ChatGPT for everybody?","Felix Wolf","2023","blog","EA Forum","forum.effectivealtruism.org/posts/esfHrQnu9aSHuXbmy/linkpost-alpaca-7b-release-or-budget-chatgpt-for-everybody",0,"",""],["Are nested jailbreaks inevitable?","judson","2023","blog","LessWrong","www.lesswrong.com/posts/mR4HaYbCpKwnJWaAu/are-nested-jailbreaks-inevitable",0,"","jailbreaks"],["GPTs are GPTs: An early look at the labor market impact potential of large language models","Tyna Eloundou and 3 others","2023","blog","openai.com","openai.com/research/gpts-are-gpts",0,"",""],["Survey on intermediate goals in AI governance","MichaelA and MaxRa","2023","blog","LessWrong","www.lesswrong.com/posts/o8fobRYGAknqdTTsM/survey-on-intermediate-goals-in-ai-governance",0,"","governance forecasting"],["Symbiosis, not alignment, as the goal for liberal democracies in the transition to artificial general intelligence","simonfriederich","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ynxDXNbmak9fEHaAm/symbiosis-not-alignment-as-the-goal-for-liberal-democracies",0,"","governance"],["Unjournal: Evaluations of \"Artificial Intelligence and Economic Growth\", and new hosting space","david_reinstein","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5d7P4gFpomfeLCHZw/unjournal-evaluations-of-artificial-intelligence-and",0,"","evals governance forecasting"],["[Appendix] Natural Abstractions: Key Claims, Theorems, and Critiques","LawrenceC and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/o7sN7moJA8TrZKtKi/appendix-natural-abstractions-key-claims-theorems-and",0,"",""],["[ASoT] Some thoughts on human abstractions","leogao","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ktJ9rCsotdqEoBtof/asot-some-thoughts-on-human-abstractions",0,"","eliciting-latent-knowledge"],["Are AI developers playing with fire?","marcusarvan","2023","blog","LessWrong","www.lesswrong.com/posts/EKc4gz7nPCkntwEwg/are-ai-developers-playing-with-fire",0,"",""],["Attribution Patching: Activation Patching At Industrial Scale","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gtLLBhzQTG6nKTeCZ/attribution-patching-activation-patching-at-industrial-scale",0,"","interpretability"],["ChatGPT getting out of the box","qbolec","2023","blog","LessWrong","www.lesswrong.com/posts/Jxp9H94trFkoene2k/chatgpt-getting-out-of-the-box",0,"",""],["Conceding a short timelines bet early","Matthew Barnett","2023","blog","LessWrong","www.lesswrong.com/posts/nto7K5W2sNR3Cpmec/conceding-a-short-timelines-bet-early",0,"","forecasting"],["Donation offsets for ChatGPT Plus subscriptions","Jeffrey Ladish","2023","blog","EA Forum","forum.effectivealtruism.org/posts/dyyXcdgBchGczruJq/donation-offsets-for-chatgpt-plus-subscriptions",0,"","governance"],["Gradual takeoff, fast failure","Max H","2023","blog","LessWrong","www.lesswrong.com/posts/3yAvb8TtLJQbAbtiC/gradual-takeoff-fast-failure",0,"","forecasting"],["Is there an analysis of the common consideration that splitting an AI lab into two (e.g. the founding of Anthropic) speeds up the development of TAI and therefore increases AI x-risk?","tchauvin","2023","blog","LessWrong","www.lesswrong.com/posts/8EZDx2GMtsTcnrPsC/is-there-an-analysis-of-the-common-consideration-that",0,"","forecasting"],["Natural Abstractions: Key claims, Theorems, and Critiques","LawrenceC and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gvzW46Z3BsaZsLc25/natural-abstractions-key-claims-theorems-and-critiques-1",0,"",""],["Privileged Bases in the Transformer Residual Stream","Nelson Elhage and 2 others","2023","blog","transformer-circuits.pub","transformer-circuits.pub/2023/privileged-basis/index.html",0,"",""],["Want to predict/explain/control the output of GPT-4? Then learn about the world, not about transformers.","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/G3tuxF4X5R5BY7fut/want-to-predict-explain-control-the-output-of-gpt-4-then",0,"",""],["We are fighting a shared battle (a call for a different approach to AI Strategy)","Gideon Futerman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Q4rg6vwbtPxXW6ECj/we-are-fighting-a-shared-battle-a-call-for-a-different",0,"","governance policy"],["What organizations other than Conjecture have (esp. public) info-hazard policies?","David Scott Krueger (formerly: capybaralet)","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qTiujsznctZcnuLF3/what-organizations-other-than-conjecture-have-esp-public",0,"",""],["AI Safety - 7 months of discussion in 17 minutes","Zoe Williams","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hCwDNq6sZofgSEN3s/ai-safety-7-months-of-discussion-in-17-minutes",0,"",""],["AI safety and consciousness research: A brainstorm","Daniel_Friedrich","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HWKqmTLcbsf4F5xAk/ai-safety-and-consciousness-research-a-brainstorm",0,"","robustness"],["ARC tests to see if GPT-4 can escape human control; GPT-4 failed to do so","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/NQ85WRcLkjnTudzdg/arc-tests-to-see-if-gpt-4-can-escape-human-control-gpt-4",0,"","evals"],["Good depictions of speed mismatches between advanced AI systems and humans?","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/g7kEqZamizsMYMGiD/good-depictions-of-speed-mismatches-between-advanced-ai",0,"","robustness"],["How well did Manifold predict GPT-4?","David Chee","2023","blog","LessWrong","www.lesswrong.com/posts/DaaFce3hBoEzYhdvz/how-well-did-manifold-predict-gpt-4",0,"","forecasting"],["Shutting Down the Lightcone Offices","Habryka and Ben Pace","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rsnrpvKofps5Py7di/shutting-down-the-lightcone-offices",0,"",""],["Towards understanding-based safety evaluations","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uqAdqrvxqGqeBHjTP/towards-understanding-based-safety-evaluations",0,"","evals"],["2023 Open Philanthropy AI Worldviews Contest: Odds of Artificial General Intelligence by 2043","srhoades10","2023","blog","EA Forum","forum.effectivealtruism.org/posts/P2AKkH8nsK8xBJ9se/2023-open-philanthropy-ai-worldviews-contest-odds-of",0,"","forecasting"],["A better analogy and example for teaching AI takeover: the ML Inferno","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/4NB5dqbjnW5imqfwq/a-better-analogy-and-example-for-teaching-ai-takeover-the-ml",0,"",""],["Comments on OpenAI’s \"Planning for AGI and beyond\"","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/03/14/comments-on-openais-planning-for-agi-and-beyond/",0,"",""],["Eliciting Latent Predictions from Transformers with the Tuned Lens","Nora Belrose and 7 others","2023","paper","arXiv preprint","arxiv.org/abs/2303.08112",0,"",""],["Fixed points in mortal population games","ViktoriaMalyasova","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HNnRCPe2CejfupSow/fixed-points-in-mortal-population-games",0,"","agents theory"],["GPT can write Quines now (GPT-4)","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ux93sLHcqmBfsRTvg/gpt-can-write-quines-now-gpt-4",0,"",""],["GPT-4 is out: thread (& links)","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eAaeeuEd4j6oJ3Ep5/gpt-4-is-out-thread-and-links",0,"",""],["Human preferences as RL critic values - implications for alignment","Seth Herd","2023","blog","LessWrong","www.lesswrong.com/posts/HEonwwQLhMB9fqABh/human-preferences-as-rl-critic-values-implications-for",0,"","rlhf"],["Storytelling Makes GPT-3.5 Deontologist: Unexpected Effects of Context on LLM Behavior","Edmund Mills and Scott Emmons","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4qC7FfCHrdaFesPzC/storytelling-makes-gpt-3-5-deontologist-unexpected-effects",0,"",""],["What is a definition, how can it be extrapolated?","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/AdyqGnvhdqDMYJaug/what-is-a-definition-how-can-it-be-extrapolated",0,"",""],["Yudkowsky on AGI risk on the Bankless podcast","Rob Bensinger","2023","blog","intelligence.org","intelligence.org/2023/03/14/yudkowsky-on-agi-risk-on-the-bankless-podcast/",0,"",""],["\"Can We Survive Technology?\" by John von Neumann","Eli Rose","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9piqRDGX6BisdMdRw/can-we-survive-technology-by-john-von-neumann",0,"",""],["Could Roko's basilisk acausally bargain with a paperclip maximizer?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/kkcQdR63LvoRZwutY/could-roko-s-basilisk-acausally-bargain-with-a-paperclip",0,"","theory"],["Discussion with Nate Soares on a key alignment difficulty","HoldenKarnofsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iy2o4nQj9DnQD7Yhj/discussion-with-nate-soares-on-a-key-alignment-difficulty",0,"",""],["Foundations for a Longtermist Foreign Policy","Manuel Carranza","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LMvFekp6tWSN9K6Em/foundations-for-a-longtermist-foreign-policy",0,"","governance policy"],["Import AI 320: Facebook's AI Lab Leak; open source ChatGPT clone; Google makes a universal translator.","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-320-facebooks-ai-lab-leak",0,"",""],["On taking AI risk seriously","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pKG5fsfrgDSQtssfu/on-taking-ai-risk-seriously",0,"",""],["On taking AI risk seriously","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/dpNkK3LJBLtaJfAvu/on-taking-ai-risk-seriously",0,"",""],["Plan for mediocre alignment of brain-like [model-based RL] AGI","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hi7zurzkCog336EC2/plan-for-mediocre-alignment-of-brain-like-model-based-rl-agi",0,"",""],["What Discovering Latent Knowledge Did and Did Not Find","Fabien Roger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bWxNPMy5MhPnQTzKz/what-discovering-latent-knowledge-did-and-did-not-find-4",0,"","eliciting-latent-knowledge"],["your terminal values are complex and not objective","Tamsin Leake","2023","blog","carado.moe","carado.moe/values-complex-not-objective.html",0,"",""],["Yudkowsky on AGI risk on the Bankless podcast","RobBensinger","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GhmcdwdT98PE5vCS2/yudkowsky-on-agi-risk-on-the-bankless-podcast",0,"",""],["An AI risk argument that resonates with NYTimes readers","Julian Bradshaw","2023","blog","LessWrong","www.lesswrong.com/posts/FzhedhEFAcKJZkgJS/an-ai-risk-argument-that-resonates-with-nytimes-readers",0,"",""],["Are there cognitive realms?","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NvwjExA7FcPDoo3L7/are-there-cognitive-realms",0,"",""],["Paper Replication Walkthrough: Reverse-Engineering Modular Addition","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Ghvdb2iwLAyGT6A3/paper-replication-walkthrough-reverse-engineering-modular",0,"","interpretability"],["the quantum amplitude argument against ethics deduplication","Tamsin Leake","2023","blog","carado.moe","carado.moe/quantum-amplitude-deduplication.html",0,"",""],["Thoughts on self-inspecting neural networks.","Deruwyn","2023","blog","LessWrong","www.lesswrong.com/posts/MwEn2e6PzzEHzsgbD/thoughts-on-self-inspecting-neural-networks",0,"","eliciting-latent-knowledge"],["[Linkpost] Scott Alexander reacts to OpenAI's latest post","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/nALdMXkxkLzysKtzC/linkpost-scott-alexander-reacts-to-openai-s-latest-post",0,"","governance"],["Compositional language for hypotheses about computations","Vanessa Kosoy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jxE47v3aezEYaydg9/compositional-language-for-hypotheses-about-computations",0,"","agents theory"],["problems for formal alignment","Tamsin Leake","2023","blog","carado.moe","carado.moe/formal-alignment-problems.html",0,"",""],["Understanding and controlling a maze-solving policy network","TurnTrout and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cAC4AXiNC5ig6jQnc/understanding-and-controlling-a-maze-solving-policy-network",0,"","interpretability policy"],["Announcing the Open Philanthropy AI Worldviews Contest","Jason Schukraft and Peter Favaloro","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NZz3Das7jFdCBN9zH/announcing-the-open-philanthropy-ai-worldviews-contest",0,"","forecasting"],["Everything's normal until it's not","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2hduXN5MXCZPqKjSv/everything-s-normal-until-it-s-not",0,"","forecasting"],["Everything's normal until it's not","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/MpD8bR9A8BFswxNq3/everything-s-normal-until-it-s-not",0,"",""],["Japan AI Alignment Conference","Chris Scammell and Katrina Joslin","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/tAQRxccEDYZY5vxvy/japan-ai-alignment-conference",0,"",""],["Japan AI Alignment Conference","ChrisScammell","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gZNSTrDD7agjo7Nqy/japan-ai-alignment-conference",0,"",""],["Operationalizing timelines","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/Nt8yDxkiMF8YAsNYA/operationalizing-timelines",0,"","forecasting"],["Reflections On The Feasibility Of Scalable-Oversight","Felix Hofstätter","2023","blog","LessWrong","www.lesswrong.com/posts/8yimdZcEWSKkutHhZ/reflections-on-the-feasibility-of-scalable-oversight",0,"","rlhf"],["Stop calling it \"jailbreaking\" ChatGPT","Templarrr","2023","blog","LessWrong","www.lesswrong.com/posts/TkKMv5xsbQ6AZ5cgD/stop-calling-it-jailbreaking-chatgpt",0,"","jailbreaks"],["Thoughts on the OpenAI alignment plan: will AI research assistants be net-positive for AI existential risk?","Jeffrey Ladish","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gt6fPgRdEHJSLGd3N/thoughts-on-the-openai-alignment-plan-will-ai-research",0,"",""],["A Roundtable for Safe AI (RSAI)?","Lara_TH","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7jbrN4aenCM9EZqyT/a-roundtable-for-safe-ai-rsai",0,"",""],["A Windfall Clause for CEO could worsen AI race dynamics","Larks","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ewroS7tsqhTsstJ44/a-windfall-clause-for-ceo-could-worsen-ai-race-dynamics",0,"","governance policy"],["Anthropic's Core Views on AI Safety","Zac Hatfield-Dodds","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/xhKr5KtvdJRssMeJ3/anthropic-s-core-views-on-ai-safety",0,"",""],["Anthropic: Core Views on AI Safety: When, Why, What, and How","jonmenaster","2023","blog","LessWrong","www.lesswrong.com/posts/MLDXcEBJ2jX7BcSmN/anthropic-core-views-on-ai-safety-when-why-what-and-how",0,"","governance"],["Challenge: construct a Gradient Hacker","Thomas Larsen and Thomas Kwa","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QvvFRDG6SG3xZ8ELz/challenge-construct-a-gradient-hacker",0,"",""],["How bad a future do ML researchers expect?","Katja Grace","2023","blog","aiimpacts.org","aiimpacts.org/how-bad-a-future-do-ml-researchers-expect/",0,"",""],["Near-term motivation for AI alignment","Victoria Krakovna","2023","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2023/03/09/near-term-motivation-for-ai-alignment/",0,"",""],["Paper Summary: The Effectiveness of AI Existential Risk Communication to the American and Dutch Public","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/fqXLT7NHZGsLmjH4o/paper-summary-the-effectiveness-of-ai-existential-risk",0,"","governance"],["QACI blobs and interval illustrated","Tamsin Leake","2023","blog","carado.moe","carado.moe/qaci-blobs-interval-illustrated.html",0,"",""],["The Translucent Thoughts Hypotheses and Their Implications","Fabien Roger","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/r3xwHzMmMf25peeHE/the-translucent-thoughts-hypotheses-and-their-implications",0,"","interpretability chain-of-thought-faithfulness"],["Utility uncertainty vs. expected information gain","michaelcohen","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Pkr97mB9Y4rkx5DdZ/utility-uncertainty-vs-expected-information-gain",0,"",""],["Why Not Just Outsource Alignment Research To An AI?","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3gAccKDW6nRKFumpP/why-not-just-outsource-alignment-research-to-an-ai",0,"",""],["[Crosspost] Why Uncontrollable AI Looks More Likely Than Ever","Otto","2023","blog","EA Forum","forum.effectivealtruism.org/posts/DpWhZaGLA5X6p5dgP/crosspost-why-uncontrollable-ai-looks-more-likely-than-ever-1",0,"",""],["Against LLM Reductionism","Erich_Grunewald","2023","blog","LessWrong","www.lesswrong.com/posts/PwfwZ2LeoLC4FXyDA/against-llm-reductionism",0,"","interpretability"],["AI Safety in a World of Vulnerable Machine Learning Systems","AdamGleave and EuanMcLean","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ncsxcf8CkDveXBCrA/ai-safety-in-a-world-of-vulnerable-machine-learning-systems-1",0,"","robustness"],["QACI blob location: no causality & answer signature","Tamsin Leake","2023","blog","carado.moe","carado.moe/blob-location.html",0,"",""],["Squeezing foundations research assistance out of formal logic narrow AI.","Donald Hobson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CknHb67jutFfBwWz3/squeezing-foundations-research-assistance-out-of-formal",0,"","robustness theory"],["Why Uncontrollable AI Looks More Likely Than Ever","otto.barten and Roman_Yampolskiy","2023","blog","LessWrong","www.lesswrong.com/posts/YEeN2yLZiMPz9Yker/why-uncontrollable-ai-looks-more-likely-than-ever",0,"","governance"],["[Linkpost] Some high-level thoughts on the DeepMind alignment team's strategy","Vika and Rohin Shah","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/a9SPcZ6GXAg9cNKdi/linkpost-some-high-level-thoughts-on-the-deepmind-alignment",0,"",""],["Alignment works both ways","Karl von Wendt","2023","blog","LessWrong","www.lesswrong.com/posts/Dx6kkXykErmAswuvS/alignment-works-both-ways",0,"",""],["Introducing AI Alignment Inc., a California public benefit corporation...","TherapistAI","2023","blog","LessWrong","www.lesswrong.com/posts/Kw8vaTc7FwaLZuyi4/introducing-ai-alignment-inc-a-california-public-benefit",0,"","automated-alignment-research"],["Should people get neuroscience phD to work in AI safety field?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/RiCueLbjrpsotmqku/should-people-get-neuroscience-phd-to-work-in-ai-safety",0,"",""],["What‘s in your list of unsolved problems in AI alignment?","jacquesthibs","2023","blog","LessWrong","www.lesswrong.com/posts/EPAofvLzsCwqYnekj/what-s-in-your-list-of-unsolved-problems-in-ai-alignment",0,"",""],["before the sharp left turn: what wins first?","Tamsin Leake","2023","blog","carado.moe","carado.moe/sharp-left-turn-what-wins-first.html",0,"",""],["Import AI 319: Sovereign AI; Facebook's weights leak on torrent networks; Google might have made a better optimizer than Adam!","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/import-ai-319-sovereign-ai-facebooks",0,"",""],["Introducing Leap Labs, an AI interpretability startup","Jessica Rumbelow","2023","blog","LessWrong","www.lesswrong.com/posts/Q44QjdtKtSoqRKgRe/introducing-leap-labs-an-ai-interpretability-startup",0,"","interpretability"],["Model-Based Policy Analysis under Deep Uncertainty","Max Reddel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kCBQHWqbk4Nrns8P7/model-based-policy-analysis-under-deep-uncertainty",0,"","policy forecasting"],["A concerning observation from media coverage of AI industry dynamics","Justin Olive","2023","blog","LessWrong","www.lesswrong.com/posts/wm9ouJPytJ9FLj3gx/a-concerning-observation-from-media-coverage-of-ai-industry",0,"","governance"],["Do humans derive values from fictitious imputed coherence?","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CBHpzpzJy98idiSGs/do-humans-derive-values-from-fictitious-imputed-coherence",0,"",""],["EA Infosec: skill up in or make a transition to infosec via this book club","Jason Clinton and Wim van der Schoot","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zxrBi4tzKwq2eNYKm/ea-infosec-skill-up-in-or-make-a-transition-to-infosec-via",0,"",""],["QACI: the problem of blob location, causality, and counterfactuals","Tamsin Leake","2023","blog","carado.moe","carado.moe/blob-causality.html",0,"",""],["Research proposal: Leveraging Jungian archetypes to create values-based models","MiguelDev","2023","blog","LessWrong","www.lesswrong.com/posts/YiSLdjyBuD3oScDxR/research-proposal-leveraging-jungian-archetypes-to-create-1",0,"",""],["Who Aligns the Alignment Researchers?","Ben Smith","2023","blog","LessWrong","www.lesswrong.com/posts/PaWpTPkbnkGRtDrDs/who-aligns-the-alignment-researchers",0,"","governance"],["Why Not Just... Build Weak AI Tools For AI Alignment Research?","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KQfYieur2DFRZDamd/why-not-just-build-weak-ai-tools-for-ai-alignment-research",0,"",""],["Contra \"Strong Coherence\"","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/AdGo5BRCzzsdDGM6H/contra-strong-coherence",0,"","agents theory"],["How to navigate potential infohazards","more better","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CnHiumpJdRmQyAfAA/how-to-navigate-potential-infohazards",0,"",""],["Misalignment Museum opens in San Francisco: ‘Sorry for killing most of humanity’","Michael Huang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZyjARuFsDBTFXeMP4/misalignment-museum-opens-in-san-francisco-sorry-for-killing",0,"",""],["More money with less risk: sell services instead of model access","lukehmiles","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pHaPds4SqfewLrEbW/more-money-with-less-risk-sell-services-instead-of-model",0,"",""],["The Benefits of Distillation in Research","Jonas Hallgren","2023","blog","EA Forum","forum.effectivealtruism.org/posts/roYj4ijkKCotSk4ob/the-benefits-of-distillation-in-research",0,"",""],["A reply to Byrnes on the Free Energy Principle","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8Be6ZQvzhsigt5fkk/a-reply-to-byrnes-on-the-free-energy-principle",0,"",""],["Acausal normalcy","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3RSq3bfnzuL3sp46J/acausal-normalcy",0,"",""],["Acausal normalcy","Andrew Critch","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Tm2wtkQkrSwEvtMAn/acausal-normalcy",0,"","theory"],["AI Governance & Strategy: Priorities, talent gaps, & opportunities","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/hAnKgips7kPyxJRY3/ai-governance-and-strategy-priorities-talent-gaps-and",0,"","governance"],["Aspiring AI safety researchers should ~argmax over AGI timelines","Ryan Kidd","2023","blog","LessWrong","www.lesswrong.com/posts/khKBJkk6T9hAQ69zT/aspiring-ai-safety-researchers-should-argmax-over-agi",0,"","forecasting"],["ChatGPT tells stories, and a note about reverse engineering: A Working Paper","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/PpdFFtxsPQK5dk4EB/chatgpt-tells-stories-and-a-note-about-reverse-engineering-a",0,"","interpretability"],["How popular is ChatGPT? Part 2: slower growth than Pokémon GO","richardkorzekwa","2023","blog","aiimpacts.org","aiimpacts.org/how-popular-is-chatgpt-part-2-slower-growth-than-pokemon-go/",0,"",""],["Introducing the new Riesgos Catastróficos Globales team","Jaime Sevilla and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sH9i6PSsXZABM5RNq/introducing-the-new-riesgos-catastroficos-globales-team",0,"","robustness"],["New Artificial Intelligence quiz: can you beat ChatGPT?","AndreFerretti","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yPvpKz7RkiS4cqKku/new-artificial-intelligence-quiz-can-you-beat-chatgpt",0,"",""],["Robin Hanson’s latest AI risk position statement","Liron","2023","blog","LessWrong","www.lesswrong.com/posts/AL6DRuE8s4yLn3yBo/robin-hanson-s-latest-ai-risk-position-statement",0,"",""],["Situational awareness in Large Language Models","Simon Möller","2023","blog","LessWrong","www.lesswrong.com/posts/TBLv9T7rAzmawehnq/situational-awareness-in-large-language-models",0,"","situational-awareness"],["state of my alignment research, and what needs work","Tamsin Leake","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RBcKeY8B5mvxiCN37/state-of-my-alignment-research-and-what-needs-work",0,"",""],["state of my alignment research, and what needs work","Tamsin Leake","2023","blog","carado.moe","carado.moe/state-research-agenda.html",0,"",""],["The Waluigi Effect (mega-post)","Cleo Nardo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/D7PumeYTDPfBTp3i7/the-waluigi-effect-mega-post",0,"","rlhf alignment-faking deception power-seeking"],["Why are counterfactuals elusive?","Martín Soto","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gwdwukkc8NfpyPitw/why-are-counterfactuals-elusive-2",0,"",""],["Call to demand answers from Anthropic about joining the AI race","sergia","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bmfR73qjHQnACQaFC/call-to-demand-answers-from-anthropic-about-joining-the-ai",0,"",""],["Don't Jump or I'll...","Double","2023","blog","LessWrong","www.lesswrong.com/posts/XvDboZ7SDBefqJwtf/don-t-jump-or-i-ll",0,"","theory"],["Game theory work on AI alignment with diverse AI systems, human individuals, & human groups?","Geoffrey Miller","2023","blog","EA Forum","forum.effectivealtruism.org/posts/XBxfmpjiWeDuhMoCJ/game-theory-work-on-ai-alignment-with-diverse-ai-systems",0,"",""],["Joscha Bach on Synthetic Intelligence [annotated]","Roman Leventov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wQERLNFoMidffTLar/joscha-bach-on-synthetic-intelligence-annotated",0,"",""],["Payor's Lemma in Natural Language","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/iCDBQtby4L2fZ7yns/payor-s-lemma-in-natural-language",0,"","theory"],["Scoring forecasts from the 2016 “Expert Survey on Progress in AI”","Harlan Stewart","2023","blog","aiimpacts.org","aiimpacts.org/scoring-forecasts-from-the-2016-expert-survey-on-progress-in-ai/",0,"","forecasting"],["The View from 30,000 Feet: Preface to the Second EleutherAI Retrospective","Stella Biderman and 3 others","2023","blog","blog.eleuther.ai","blog.eleuther.ai/year-two-preface/",0,"",""],["What are some sources related to big-picture AI strategy?","Jacob_Watts","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9PNya9tF3bKPbLoGf/what-are-some-sources-related-to-big-picture-ai-strategy",0,"","governance"],["Call for Cruxes by Rhyme, a Longtermist History Consultancy","Lara_TH","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hfXy8EbyNTuBixjJf/call-for-cruxes-by-rhyme-a-longtermist-history-consultancy",0,"","governance forecasting"],["Call for Cruxes by Rhyme, a Longtermist History Consultancy","Lara","2023","blog","LessWrong","www.lesswrong.com/posts/pwc4TDQPtRvC3mCsR/call-for-cruxes-by-rhyme-a-longtermist-history-consultancy",0,"","governance"],["Existential Risk from Power-Seeking AI","Joe Carlsmith","2023","report","jc.gatspress.com","jc.gatspress.com/pdf/existential_risk_and_powerseeking_ai.pdf",0,"","power-seeking"],["Extreme GDP growth is a bad operating definition of \"slow takeoff\"","lc","2023","blog","LessWrong","www.lesswrong.com/posts/sTNuKcF63s9SneDPT/extreme-gdp-growth-is-a-bad-operating-definition-of-slow",0,"","forecasting"],["Implied \"utilities\" of simulators are broad, dense, and shallow","porby","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/k48vB92mjE9Z28C3s/implied-utilities-of-simulators-are-broad-dense-and-shallow",0,"",""],["Inside the mind of a superhuman Go model: How does Leela Zero read ladders?","Haoxing Du","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FF8i6SLfKb4g7C4EL/inside-the-mind-of-a-superhuman-go-model-how-does-leela-zero-2",0,"","interpretability"],["Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 Small","Kevin Ro Wang and 4 others","2023","report","openreview.net","openreview.net/pdf?id=NpsVSN6o4ul",0,"","interpretability mechanistic-interpretability"],["on strong/general coherent agents","Tamsin Leake","2023","blog","carado.moe","carado.moe/strongly-generally-coherent-agents.html",0,"","agents"],["Predictions for shard theory mechanistic interpretability results","TurnTrout and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JusJcepE2qohiC3hm/predictions-for-shard-theory-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability forecasting"],["Problems of people new to AI safety and my project ideas to mitigate them","Igor Ivanov","2023","blog","LessWrong","www.lesswrong.com/posts/jGW3FwkpFdsjrpMe5/problems-of-people-new-to-ai-safety-and-my-project-ideas-to",0,"",""],["Scoring forecasts from the 2016 “Expert Survey on Progress in AI”","PatrickL","2023","blog","LessWrong","www.lesswrong.com/posts/tQwjkFT8s2uf2arFN/scoring-forecasts-from-the-2016-expert-survey-on-progress-in",0,"","forecasting"],["Some Variants of Sleeping Beauty","Sylvester Kollin and Eric Chen","2023","blog","LessWrong","www.lesswrong.com/posts/hA5FvFaajX7fvwKjZ/some-variants-of-sleeping-beauty",0,"","theory"],["Taboo \"compute overhang\"","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/icR53xeAkeuzgzsWP/taboo-compute-overhang",0,"","forecasting"],["$20 Million in NSF Grants for Safety Research","Dan H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jwe6jpubuMiuSRqff/usd20-million-in-nsf-grants-for-safety-research",0,"",""],["A mostly critical review of infra-Bayesianism","matolcsid","2023","blog","LessWrong","www.lesswrong.com/posts/StkjjQyKwg7hZjcGB/a-mostly-critical-review-of-infra-bayesianism",0,"","agents theory"],["Do you worry about totalitarian regimes using AI Alignment technology to create AGI that subscribe to their values?","diodio_yang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hy2qcaYStNTBqaZCs/do-you-worry-about-totalitarian-regimes-using-ai-alignment",0,"",""],["Heuristics on bias to action versus status quo?","Farkas","2023","blog","LessWrong","www.lesswrong.com/posts/s2z6hKbzAyuPKeset/heuristics-on-bias-to-action-versus-status-quo",0,"","theory"],["Performance guarantees in classical learning theory and infra-Bayesianism","matolcsid","2023","blog","LessWrong","www.lesswrong.com/posts/q6dQpSfNHCYzKb2mf/performance-guarantees-in-classical-learning-theory-and",0,"","agents theory"],["Power-seeking can be probable and predictive for trained agents","Vika and janos","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fLpuusx9wQyyEBtkJ/power-seeking-can-be-probable-and-predictive-for-trained",0,"","power-seeking agents"],["Scarce Channels and Abstraction Coupling","johnswentworth","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/CvibiLyHj3n3Aigez/scarce-channels-and-abstraction-coupling",0,"",""],["Some Things I Heard about AI Governance at EAG","utilistrutil","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iqDt8YFLjvtjBPyv6/some-things-i-heard-about-ai-governance-at-eag",0,"","governance"],["Transcript: Testing ChatGPT's Performance in Engineering","alxgoldstn","2023","blog","LessWrong","www.lesswrong.com/posts/RsvKae3KSXEaivpgJ/transcript-testing-chatgpt-s-performance-in-engineering",0,"","forecasting"],["What does Bing Chat tell us about AI risk?","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/what-does-bing-chat-tell-us-about-ai-risk/",0,"",""],["What does Bing Chat tell us about AI risk?","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Q9tiLjgdHTMqFYsii/what-does-bing-chat-tell-us-about-ai-risk",0,"",""],["[Simulators seminar sequence] #2 Semiotic physics - revamped","Jan and 9 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/TTn6vTcZ3szBctvgb/simulators-seminar-sequence-2-semiotic-physics-revamped",0,"",""],["Counting-down vs. counting-up coherence","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/svpnmmeJresYs23rY/counting-down-vs-counting-up-coherence",0,"",""],["Seeking input on a list of AI books for broader audience","Darren McKee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BxgwGYFuKFu5ioBjs/seeking-input-on-a-list-of-ai-books-for-broader-audience",0,"",""],["Some thoughts pointing to slower AI take-off","Bastiaan","2023","blog","LessWrong","www.lesswrong.com/posts/Ci9FJcXtgnmSDMkyu/some-thoughts-pointing-to-slower-ai-take-off",0,"","forecasting"],["The idea of an \"aligned superintelligence\" seems misguided","ssadler","2023","blog","LessWrong","www.lesswrong.com/posts/xn59deCXbe99EoG86/the-idea-of-an-aligned-superintelligence-seems-misguided",0,"",""],["Why I think it's important to work on AI forecasting","Matthew_Barnett","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zrSx3NRZEaJENazHK/why-i-think-it-s-important-to-work-on-ai-forecasting",0,"","forecasting"],["[Link Post] Cyber Digital Authoritarianism (National Intelligence Council Report)","Phosphorous","2023","blog","LessWrong","www.lesswrong.com/posts/DfcGHqgxAWAL6BCst/link-post-cyber-digital-authoritarianism-national",0,"","governance"],["A library for safety research in conditioning on RLHF tasks","James Chua","2023","blog","LessWrong","www.lesswrong.com/posts/aHPmGPWtmK259J8ou/a-library-for-safety-research-in-conditioning-on-rlhf-tasks",0,"","rlhf"],["A mechanistic explanation for SolidGoldMagikarp-like tokens in GPT2","MadHatter","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dFbfCLZA4pejckeKc/a-mechanistic-explanation-for-solidgoldmagikarp-like-tokens",0,"",""],["An economics of AI gov - best resources for","Liv","2023","blog","EA Forum","forum.effectivealtruism.org/posts/oQyMgPu8pQbCKxHjQ/an-economics-of-ai-gov-best-resources-for",0,"","governance policy"],["How to ‘troll for good’: Leveraging IP for AI governance","Michael Huang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sC69HBdkLuq58Yzpw/how-to-troll-for-good-leveraging-ip-for-ai-governance",0,"","governance policy"],["Incentives and Selection: A Missing Frame From AI Threat Discussions?","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/5HvCSt3vDSDuuXkfz/incentives-and-selection-a-missing-frame-from-ai-threat",0,"","alignment-faking deception"],["some thoughts about terminal alignment","Tamsin Leake","2023","blog","carado.moe","carado.moe/terminal-alignment-solutions.html",0,"",""],["The Preference Fulfillment Hypothesis","Kaj_Sotala","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kf6sKZudduhJmykTg/the-preference-fulfillment-hypothesis",0,"",""],["Very Briefly: The CHIPS Act","Yadav","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WKSwH4eyDiqhJMcrz/very-briefly-the-chips-act-1",0,"","governance policy"],["clarifying formal alignment implementation","Tamsin Leake","2023","blog","carado.moe","carado.moe/clarifying-formal-alignment-implementation.html",0,"",""],["Cognitive Emulation: A Naive AI Safety Proposal","Connor Leahy and Gabriel Alfour","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ngEvKav9w57XrGQnb/cognitive-emulation-a-naive-ai-safety-proposal",0,"",""],["Which is more important for reducing s-risks, researching on AI sentience or animal welfare?","jackchang110","2023","blog","EA Forum","forum.effectivealtruism.org/posts/u8hC6LkEqw4xaJqyh/which-is-more-important-for-reducing-s-risks-researching-on",0,"",""],["Would more model evals teams be good?","Ryan Kidd","2023","blog","LessWrong","www.lesswrong.com/posts/fiqDjKCGzaopWoz7s/would-more-model-evals-teams-be-good",0,"","evals governance"],["2023 Stanford Existential Risks Conference","elizabethcooper","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ukszSQHPMN4kyyKRx/2023-stanford-existential-risks-conference",0,"","policy"],["Agents vs. Predictors: Concrete differentiating factors","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/eQ4eLQAmPvp9anJcB/agents-vs-predictors-concrete-differentiating-factors",0,"","agents"],["Christiano (ARC) and GA (Conjecture) Discuss Alignment Cruxes","Andrea_Miotti and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pgpFHLJnv7AdSi3qS/christiano-arc-and-ga-conjecture-discuss-alignment-cruxes",0,"",""],["How major governments can help with the most important century","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/how-governments-can-help-with-the-most-important-century/",0,"",""],["How major governments can help with the most important century","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ruJnXtdDS7XiiwzSP/how-major-governments-can-help-with-the-most-important",0,"","evals governance"],["How popular is ChatGPT? Part 1: more popular than Taylor Swift","Harlan Stewart","2023","blog","aiimpacts.org","aiimpacts.org/how-popular-is-chatgpt-part-1-more-popular-than-taylor-swift/",0,"",""],["Meta \"open sources\" LMs competitive with Chinchilla, PaLM, and code-davinci-002 (Paper)","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ChbRgvuGaG2dAtr6i/meta-open-sources-lms-competitive-with-chinchilla-palm-and",0,"",""],["Retrospective on the 2022 Conjecture AI Discussions","Andrea_Miotti","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BEyAWbCdtWpSGxmun/retrospective-on-the-2022-conjecture-ai-discussions",0,"",""],["Sam Altman: \"Planning for AGI and beyond\"","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/zRn6aQyD8uhAN7qCc/sam-altman-planning-for-agi-and-beyond",0,"",""],["Training for corrigability: obvious problems?","Ben Amitay","2023","blog","LessWrong","www.lesswrong.com/posts/FgHBatgKyYR6oqFz5/training-for-corrigability-obvious-problems",0,"",""],["AI that shouldn't work, yet kind of does","Donald Hobson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NK2CeDNKMEY9gRZp2/ai-that-shouldn-t-work-yet-kind-of-does",0,"",""],["Automated Sandwiching & Quantifying Human-LLM Cooperation: ScaleOversight hackathon results","Esben Kran and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/qwQqZtjWdyDLC4JTB/automated-sandwiching-and-quantifying-human-llm-cooperation",0,"","governance"],["EIS XII: Summary","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aymbce8ge9ve2C4Po/eis-xii-summary",0,"","interpretability robustness"],["EIS XII: Summary","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/aymbce8ge9ve2C4Po/eis-xii-summary",0,"","interpretability robustness"],["Full Transcript: Eliezer Yudkowsky on the Bankless podcast","remember and Andrea_Miotti","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Aq82XqYhgqdPdPrBA/full-transcript-eliezer-yudkowsky-on-the-bankless-podcast",0,"",""],["Hello, Elua.","Tamsin Leake","2023","blog","carado.moe","carado.moe/hello-elua.html",0,"",""],["Searching for a model's concepts by their shape – a theoretical framework","Kaarel and 5 others","2023","blog","LessWrong","www.lesswrong.com/posts/Go5ELsHAyw7QrArQ6/searching-for-a-model-s-concepts-by-their-shape-a",0,"","interpretability eliciting-latent-knowledge"],["Taking a leave of absence from Open Philanthropy to work on AI safety","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aJwcgm2nqiZu6zq2S/taking-a-leave-of-absence-from-open-philanthropy-to-work-on",0,"",""],["Teleosemantics!","abramdemski","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WGEPBmErv8ufrq8Fc/teleosemantics",0,"",""],["Cyborg Periods: There will be multiple AI transitions","Jan_Kulveit and rosehadshar","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BTApNmv7s6RTGxeP4/cyborg-periods-there-will-be-multiple-ai-transitions",0,"","automated-alignment-research governance forecasting"],["EIS XI: Moving Forward","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/L5Rua9aTndviy8dvc/eis-xi-moving-forward",0,"","interpretability"],["Intervening in the Residual Stream","MadHatter","2023","blog","LessWrong","www.lesswrong.com/posts/pYummGFJu3Jsoja8B/intervening-in-the-residual-stream",0,"","interpretability"],["Is there a ML agent that abandons it's utility function out-of-distribution without losing capabilities?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/mL89Ze5uZTX3udb3a/is-there-a-ml-agent-that-abandons-it-s-utility-function-out",0,"","scalable-oversight agents robustness"],["Power-Seeking = Minimising free energy","Jonas Hallgren","2023","blog","LessWrong","www.lesswrong.com/posts/KYxpkoh8ppnPfmuF3/power-seeking-minimising-free-energy",0,"","power-seeking"],["The Open Agency Model","Eric Drexler","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5hApNw5f7uG8RXxGS/the-open-agency-model",0,"",""],["The shallow reality of 'deep learning theory'","Jesse Hoogland","2023","blog","LessWrong","www.lesswrong.com/posts/BDTfddkttFXHqGnEi/the-shallow-reality-of-deep-learning-theory",0,"","interpretability"],["Video/animation: Neel Nanda explains what mechanistic interpretability is","DanielFilan","2023","blog","LessWrong","www.lesswrong.com/posts/JtrqxAae5JAFthkkb/video-animation-neel-nanda-explains-what-mechanistic",0,"","interpretability mechanistic-interpretability"],["[Preprint] Pretraining Language Models with Human Preferences","Giulio","2023","blog","LessWrong","www.lesswrong.com/posts/mNj6eqd95Csv8kX3f/preprint-pretraining-language-models-with-human-preferences",0,"","rlhf"],["A proof of inner Löb's theorem","James Payor","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QgZAbFHtgSGjx4aTS/a-proof-of-inner-loeb-s-theorem",0,"",""],["Basic facts about language models during training","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2JJtxitp6nqu6ffak/basic-facts-about-language-models-during-training-1",0,"","interpretability"],["Breaking the Optimizer’s Curse, and Consequences for Existential Risks and Value Learning","Roger Dearnaley","2023","blog","LessWrong","www.lesswrong.com/posts/ZqTQtEvBQhiGy6y7p/breaking-the-optimizer-s-curse-and-consequences-for-1",0,"",""],["Deceptive Alignment is <1% Likely by Default","DavidW","2023","blog","LessWrong","www.lesswrong.com/posts/RTkatYxJWvXR4Qbyd/deceptive-alignment-is-less-than-1-likely-by-default",0,"","alignment-faking deception"],["Does most of your impact come from what you do soon?","Joshc","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4FnzE6eCEAbwTux99/does-most-of-your-impact-come-from-what-you-do-soon",0,"",""],["EIS X: Continual Learning, Modularity, Compression, and Biological Brains","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fHnwCDDbDHWqbJ8Nd/eis-x-continual-learning-modularity-compression-and",0,"","interpretability robustness"],["Instrumentality makes agents agenty","porby","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/EBKJq2gkhvdMg5nTQ/instrumentality-makes-agents-agenty",0,"","instrumental-convergence agents"],["Pretraining Language Models with Human Preferences","Tomek Korbak and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8F4dXYriqbsom46x5/pretraining-language-models-with-human-preferences",0,"","rlhf"],["What is it like doing AI safety work?","Kat Woods and peterbarnett","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yiTcjSWuy7ptTb5XS/what-is-it-like-doing-ai-safety-work",0,"",""],["You're not a simulation, 'cause you're hallucinating","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dMBmZNwdjQ6yHvWZ5/you-re-not-a-simulation-cause-you-re-hallucinating",0,"",""],["[MLSN #8] Mechanistic interpretability, using law to inform AI alignment, scaling laws for proxy gaming","Dan H and ThomasW","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/whq89vpQPp7mo5FG2/mlsn-8-mechanistic-interpretability-using-law-to-inform-ai",0,"","interpretability mechanistic-interpretability scaling-laws"],["[MLSN #8]: Mechanistic interpretability, using law to inform AI alignment, scaling laws for proxy gaming","ThomasW and Dan H","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Y7croZavYcv88Z7WK/mlsn-8-mechanistic-interpretability-using-law-to-inform-ai",0,"","interpretability mechanistic-interpretability scaling-laws"],["A circuit for Python docstrings in a 4-layer attention-only transformer","StefanHex and Jett","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/u6KXXmKFbXfWzoAXn/a-circuit-for-python-docstrings-in-a-4-layer-attention-only",0,"","interpretability mechanistic-interpretability"],["AGI doesn't need understanding, intention, or consciousness in order to kill us, only intelligence","James Blaha","2023","blog","LessWrong","www.lesswrong.com/posts/ip2vSzkcmi3rBbnYE/agi-doesn-t-need-understanding-intention-or-consciousness-in",0,"",""],["Behavioral and mechanistic definitions (often confuse AI alignment discussions)","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Si52fuEGSJJTXW9zs/behavioral-and-mechanistic-definitions-often-confuse-ai",0,"",""],["Bing finding ways to bypass Microsoft's filters without being asked. Is it reproducible?","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/hGnqS8DKQnRe43Xdg/bing-finding-ways-to-bypass-microsoft-s-filters-without",0,"",""],["Don't Call It AI Alignment","RedStateBlueState","2023","blog","EA Forum","forum.effectivealtruism.org/posts/6aYfWyo9DKEheogf8/don-t-call-it-ai-alignment",0,"",""],["EIS IX: Interpretability and Adversaries","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kYNMXjg8Tmcq3vjM6/eis-ix-interpretability-and-adversaries",0,"","interpretability robustness"],["Emergent Deception and Emergent Optimization","jsteinhardt","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aEjckcqHZZny9L2zy/emergent-deception-and-emergent-optimization",0,"","deception"],["ML Safety Newsletter #8","Dan Hendrycks","2023","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-8",0,"",""],["There are no coherence theorems","Dan H and EJT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/yCuzmCsE86BTu9PfA/there-are-no-coherence-theorems",0,"",""],["There are no coherence theorems","EJT","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FoRyordtA7LDoEhd7/there-are-no-coherence-theorems",0,"","theory"],["Validator models: A simple approach to detecting goodharting","beren","2023","blog","LessWrong","www.lesswrong.com/posts/r6f9DPBZYpWFw8Qrb/validator-models-a-simple-approach-to-detecting-goodharting",0,"","rlhf goodharts-law"],["What AI companies can do today to help with the most important century","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/what-ai-companies-can-do-today-to-help-with-the-most-important-century/",0,"",""],["What AI companies can do today to help with the most important century","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/i6btyefRRX23yCpnP/what-ai-companies-can-do-today-to-help-with-the-most",0,"","evals"],["What to think when a language model tells you it's sentient","rgb","2023","blog","EA Forum","forum.effectivealtruism.org/posts/P4ut25NhfsFEMEeLJ/what-to-think-when-a-language-model-tells-you-it-s-sentient",0,"",""],["A Neural Network undergoing Gradient-based Training as a Complex System","Spencer Becker-Kahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8edorDRSbJa9TCipa/a-neural-network-undergoing-gradient-based-training-as-a",0,"",""],["Degamification","Nate Showell","2023","blog","LessWrong","www.lesswrong.com/posts/xHxTxHfMeJS5y2L3i/degamification",0,"","goodharts-law"],["Does novel understanding imply novel agency / values?","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DE58ifrwYW5ogiSyJ/does-novel-understanding-imply-novel-agency-values",0,"",""],["EIS VIII: An Engineer’s Understanding of Deceptive Alignment","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aDDjCJAGqcpmA5apw/eis-viii-an-engineer-s-understanding-of-deceptive-alignment",0,"","interpretability alignment-faking deception"],["AGI in sight: our look at the game board","Andrea_Miotti and Gabriel Alfour","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/PE22QJSww8mpwh7bt/agi-in-sight-our-look-at-the-game-board",0,"","governance"],["EIS VII: A Challenge for Mechanists","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KSHqLzQscwJnv44T8/eis-vii-a-challenge-for-mechanists",0,"","interpretability"],["Interview with Roman Yampolskiy about AGI on The Reality Check","Darren McKee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/bQnJzDaFBwrtPLR6f/interview-with-roman-yampolskiy-about-agi-on-the-reality",0,"",""],["Parametrically retargetable decision-makers tend to seek power","TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GY49CKBkEs3bEpteM/parametrically-retargetable-decision-makers-tend-to-seek",0,"","instrumental-convergence power-seeking"],["Should ChatGPT make us downweight our belief in the consciousness of non-human animals?","splinter","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Bi8av6iknHFXkSxnS/should-chatgpt-make-us-downweight-our-belief-in-the",0,"",""],["AI Safety Info Distillation Fellowship","Robert Miles and mwatkins","2023","blog","LessWrong","www.lesswrong.com/posts/arkQaWauCvkTvgcRH/ai-safety-info-distillation-fellowship",0,"",""],["Automating Consistency","Hoagy","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/maeTg6zXw4DBAXXrK/automating-consistency",0,"","chain-of-thought-faithfulness"],["EIS VI: Critiques of Mechanistic Interpretability Work in AI Safety","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wt7HXaCWzuKQipqz3/eis-vi-critiques-of-mechanistic-interpretability-work-in-ai",0,"","interpretability mechanistic-interpretability"],["How good/bad is the new Bing AI for the world?","Nathan Young","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rEMz3NwvJMATSWS5y/how-good-bad-is-the-new-bing-ai-for-the-world",0,"",""],["How should AI systems behave, and who should decide? [OpenAI blog]","ShardPhoenix","2023","blog","LessWrong","www.lesswrong.com/posts/L9AdAtuxwRwTNdhDy/how-should-ai-systems-behave-and-who-should-decide-openai",0,"","governance"],["I Am Scared of Posting Negative Takes About Bing's AI","Yitz","2023","blog","LessWrong","www.lesswrong.com/posts/xri58L7WkyeKyKv4P/i-am-scared-of-posting-negative-takes-about-bing-s-ai",0,"",""],["One-layer transformers aren’t equivalent to a set of skip-trigrams","Buck","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/b5HNYh9ne5vEkX5ag/one-layer-transformers-aren-t-equivalent-to-a-set-of-skip",0,"",""],["Powerful mesa-optimisation is already here","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GLrnyH4ChFhMqsy4v/powerful-mesa-optimisation-is-already-here",0,"","forecasting"],["The public supports regulating AI for safety","Zach Stein-Perlman","2023","blog","aiimpacts.org","aiimpacts.org/the-public-supports-regulating-ai-for-safety/",0,"",""],["Two problems with ‘Simulators’ as a frame","ryan_greenblatt","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HD2s4mj4fsx6WtFAR/two-problems-with-simulators-as-a-frame",0,"",""],["don't censor yourself, silly !","Tamsin Leake","2023","blog","carado.moe","carado.moe/dont-censor-yourself-silly.html",0,"",""],["EIS V: Blind Spots In AI Safety Interpretability Research","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/7TFJAvjYfMKxKQ4XS/eis-v-blind-spots-in-ai-safety-interpretability-research",0,"","interpretability"],["Non-Unitary Quantum Logic -- SERI MATS Research Sprint","Yegreg","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cYJqGWuBwymLdFpLT/non-unitary-quantum-logic-seri-mats-research-sprint",0,"",""],["Paper: The Capacity for Moral Self-Correction in Large Language Models (Anthropic)","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/PrLnptfNDg2wBWNyb/paper-the-capacity-for-moral-self-correction-in-large",0,"","rlhf"],["Pretraining Language Models with Human Preferences","Tomasz Korbak and 7 others","2023","paper","arXiv preprint","arxiv.org/abs/2302.08582",0,"","rlhf benchmarks unlearning"],["a narrative explanation of the QACI alignment plan","Tamsin Leake","2023","blog","carado.moe","carado.moe/narrative-explanation-qaci.html",0,"",""],["AI alignment researchers may have a comparative advantage in reducing s-risks","Lukas_Gloor","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8yaQ6i3oaFLprsFyb/ai-alignment-researchers-may-have-a-comparative-advantage-in",0,"",""],["Don't accelerate problems you're trying to solve","Andrea_Miotti and remember","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4Pi3WhFb4jPphBzme/don-t-accelerate-problems-you-re-trying-to-solve",0,"",""],["EIS IV: A Spotlight on Feature Attribution/Saliency","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/f8nd9F7dL9SxueLFA/eis-iv-a-spotlight-on-feature-attribution-saliency",0,"","interpretability"],["EIS IV: A Spotlight on Feature Attribution/Saliency","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/f8nd9F7dL9SxueLFA/eis-iv-a-spotlight-on-feature-attribution-saliency",0,"","interpretability"],["Huh. Bing thing got me real anxious about AI. Resources to help with that please?","Arvin","2023","blog","EA Forum","forum.effectivealtruism.org/posts/reWKxv7xXpwRnZJLA/huh-bing-thing-got-me-real-anxious-about-ai-resources-to",0,"",""],["Order Matters for Deceptive Alignment","DavidW","2023","blog","LessWrong","www.lesswrong.com/posts/CsjLDAhQat4PY6dsc/order-matters-for-deceptive-alignment-1",0,"","alignment-faking deception"],["The Capacity for Moral Self-Correction in Large Language Models","Deep Ganguli and 35 others","2023","paper","arXiv preprint","arxiv.org/abs/2302.07459",0,"","rlhf"],["EIS III: Broad Critiques of Interpretability Research","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/gwG9uqw255gafjYN4/eis-iii-broad-critiques-of-interpretability-research",0,"","interpretability"],["explaining \".\"","Tamsin Leake","2023","blog","carado.moe","carado.moe/explaining-dot.html",0,"",""],["Explaining SolidGoldMagikarp by looking at it from random directions","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/jbi9kxhb4iCQyWG9Y/explaining-solidgoldmagikarp-by-looking-at-it-from-random",0,"","interpretability"],["Qualities that alignment mentors value in junior researchers","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/wYEwx6xcY2JxBJsfA/qualities-that-alignment-mentors-value-in-junior-researchers",0,"",""],["SolidGoldMagikarp III: Glitch token archaeology","mwatkins and Jessica Rumbelow","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/8viQEp8KBg2QSW4Yc/solidgoldmagikarp-iii-glitch-token-archaeology",0,"",""],["The Cave Allegory Revisited: Understanding GPT's Worldview","Jan_Kulveit","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kFCu3batN8k8mwtmh/the-cave-allegory-revisited-understanding-gpt-s-worldview",0,"",""],["The Linguistic Blind Spot of Value-Aligned Agency, Natural and Artificial","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QJEmnYKJt4kMeDhfy/the-linguistic-blind-spot-of-value-aligned-agency-natural",0,"","agents"],["The Linguistic Blind Spot of Value-Aligned Agency, Natural and Artificial","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/QJEmnYKJt4kMeDhfy/the-linguistic-blind-spot-of-value-aligned-agency-natural",0,"","agents"],["Whole Bird Emulation requires Quantum Mechanics","Jeffrey Heninger","2023","blog","aiimpacts.org","aiimpacts.org/whole-bird-emulation-requires-quantum-mechanics/",0,"",""],["4 ways to think about democratizing AI [GovAI Linkpost]","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/EWeCmbMyDaTnD8Guc/4-ways-to-think-about-democratizing-ai-govai-linkpost",0,"","governance"],["is intelligence program inversion?","Tamsin Leake","2023","blog","carado.moe","carado.moe/is-intelligence-program-inversion.html",0,"",""],["LLM Basics: Embedding Spaces - Transformer Token Vectors Are Not Points in Space","NickyP","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pHPmMGEMYefk9jLeh/llm-basics-embedding-spaces-transformer-token-vectors-are",0,"","interpretability"],["Morphological intelligence, superhuman empathy, and ethical arbitration","Roman Leventov","2023","blog","LessWrong","www.lesswrong.com/posts/6EspRSzYNnv9DPhkr/morphological-intelligence-superhuman-empathy-and-ethical",0,"",""],["fuzzies & utils: check that you're getting either","Tamsin Leake","2023","blog","carado.moe","carado.moe/fuzzies-utils-check-getting-either.html",0,"",""],["High impact job opportunity at ARIA (UK)","Rasool","2023","blog","EA Forum","forum.effectivealtruism.org/posts/furE5aznZCNDjkdmb/high-impact-job-opportunity-at-aria-uk",0,"","governance policy"],["Jobs that can help with the most important century","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/njD2PurEKDEZcMLKZ/jobs-that-can-help-with-the-most-important-century",0,"",""],["The conceptual Doppelgänger problem","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pgsevroJ265WcScHu/the-conceptual-doppelgaenger-problem",0,"","interpretability"],["Why almost every RL agent does learned optimization","Lee Sharkey","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/J8ifgynkfhpmrGrL8/why-almost-every-rl-agent-does-learned-optimization",0,"","agents"],["A note on 'semiotic physics'","metasemi","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/AdXzZDoYFqHCfupDB/a-note-on-semiotic-physics",0,"",""],["GPT is dangerous because it is useful at all","Tamsin Leake","2023","blog","carado.moe","carado.moe/gpt-dangerous-useful.html",0,"",""],["my takeoff speeds? depends how you define that","Tamsin Leake","2023","blog","carado.moe","carado.moe/takeoff-speeds-define.html",0,"","forecasting"],["Shortening Timelines: There's No Buffer Anymore","Jeff Rose","2023","blog","LessWrong","www.lesswrong.com/posts/THJbo4ygsE2d5GvkP/shortening-timelines-there-s-no-buffer-anymore",0,"","forecasting"],["The Importance of AI Alignment, explained in 5 points","Daniel_Eth","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CghaRkCDKYTbMhorc/the-importance-of-ai-alignment-explained-in-5-points",0,"",""],["Threatening to do the impossible: A solution to spurious counterfactuals for functional decision theory via proof theory","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/rxDLTGZu2ExELu4ZX/threatening-to-do-the-impossible-a-solution-to-spurious",0,"","theory"],["We Found An Neuron in GPT-2","Joseph Miller and Clement Neo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cgqh99SHsCv3jJYDS/we-found-an-neuron-in-gpt-2",0,"","interpretability"],["A proposed method for forecasting transformative AI","Matthew Barnett","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4ufbirCCLsFiscWuY/a-proposed-method-for-forecasting-transformative-ai",0,"","forecasting"],["Conditioning Predictive Models: Open problems, Conclusion, and Appendix","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/a2sw7HKyjnAAp2oZ4/conditioning-predictive-models-open-problems-conclusion-and",0,"",""],["Cyborgism","NicholasKees and janus","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bxt7uCiHam4QXrQAA/cyborgism",0,"","automated-alignment-research"],["FLI Podcast: Connor Leahy on AI Progress, Chimps, Memes, and Markets (Part 1/3)","remember and Andrea_Miotti","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hfkjegoMZ8j8KGyiR/fli-podcast-connor-leahy-on-ai-progress-chimps-memes-and",0,"",""],["Jobs that can help with the most important century","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/jobs-that-can-help-with-the-most-important-century/",0,"",""],["Many important technologies start out as science fiction before becoming real","trevor","2023","blog","LessWrong","www.lesswrong.com/posts/6iDpq3GoNpfYiuBa3/many-important-technologies-start-out-as-science-fiction",0,"",""],["Mechanism Design for AI Safety - Agenda Creation Retreat","Rubi J. Hudson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/uPnmzDnoSviCcKq2L/mechanism-design-for-ai-safety-agenda-creation-retreat",0,"",""],["Why I’m not working on {debate, RRM, ELK, natural abstractions}","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/PDx4ueLpvz5gxPEus/why-i-m-not-working-on-debate-rrm-elk-natural-abstractions",0,"",""],["Anomalous tokens reveal the original identities of Instruct models","janus and jdp","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LAxAmooK4uDfWmbep/anomalous-tokens-reveal-the-original-identities-of-instruct",0,"",""],["Anomalous tokens reveal the original identities of Instruct models","janus","2023","blog","generative.ink","generative.ink/posts/anomalous-tokens-reveal-the-original-identities-of-instruct-models/",0,"",""],["Apply to the Cambridge ML for Alignment Bootcamp (CaMLAB) [26 March - 8 April]","hannah","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hwyzytrEhdDeoyPzH/apply-to-the-cambridge-ml-for-alignment-bootcamp-camlab-26",0,"",""],["CEV can be coherent enough","Tamsin Leake","2023","blog","carado.moe","carado.moe/cev-coherent-enough.html",0,"",""],["Conditioning Predictive Models: Deployment strategy","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NXdTxyWy2PEXueKwi/conditioning-predictive-models-deployment-strategy",0,"",""],["Do the Safety Properties of Powerful AI Systems Need to be Adversarially Robust? Why?","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/2ew4NFZovxCLsvHKS/do-the-safety-properties-of-powerful-ai-systems-need-to-be",0,"","goodharts-law robustness"],["EIS II: What is “Interpretability”?","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MyvkTKfndx9t4zknh/eis-ii-what-is-interpretability",0,"","interpretability"],["EIS II: What is “Interpretability”?","scasper","2023","blog","LessWrong","www.lesswrong.com/posts/MyvkTKfndx9t4zknh/eis-ii-what-is-interpretability",0,"","interpretability"],["Notes on the Mathematics of LLM Architectures","Spencer Becker-Kahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rFhxbWCdECxuT9xa2/notes-on-the-mathematics-of-llm-architectures",0,"",""],["On Developing a Mathematical Theory of Interpretability","Spencer Becker-Kahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rEMpTapcAzjTiSckf/on-developing-a-mathematical-theory-of-interpretability",0,"","interpretability"],["Security Mindset - Fire Alarms and Trigger Signatures","elspood","2023","blog","LessWrong","www.lesswrong.com/posts/meryibcAerAr5b4Hh/security-mindset-fire-alarms-and-trigger-signatures",0,"","governance"],["Speedrun: AI Alignment Prizes","joe","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SgeLEhS3zDfRBcXQG/speedrun-ai-alignment-prizes",0,"",""],["Technological developments that could increase risks from nuclear weapons: A shallow review","MichaelA and Will Aldred","2023","blog","EA Forum","forum.effectivealtruism.org/posts/HuQtr7qfB2EfcGqTu/technological-developments-that-could-increase-risks-from-1",0,"","forecasting"],["Technology is Power: Raising Awareness Of Technological Risks","Marc Wong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WdkfjWZiBnLjc7Geg/technology-is-power-raising-awareness-of-technological-risks",0,"",""],["The Engineer’s Interpretability Sequence (EIS) I: Intro","scasper","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ExRN5Bu3696cf9Ccm/the-engineer-s-interpretability-sequence-eis-i-intro",0,"","interpretability"],["Using PICT against PastaGPT Jailbreaking","Quentin FEUILLADE--MONTIXI","2023","blog","LessWrong","www.lesswrong.com/posts/WWmGEix82myHjhHYB/using-pict-against-pastagpt-jailbreaking",0,"","jailbreaks"],["A (EtA: quick) note on terminology: AI Alignment != AI x-safety","David Scott Krueger (formerly: capybaralet)","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/biP5XBmqvjopvky7P/a-eta-quick-note-on-terminology-ai-alignment-ai-x-safety",0,"",""],["A (EtA: quick) note on terminology: AI Alignment != AI x-safety","David Scott Krueger (formerly: capybaralet)","2023","blog","LessWrong","www.lesswrong.com/posts/biP5XBmqvjopvky7P/a-eta-quick-note-on-terminology-ai-alignment-ai-x-safety",0,"",""],["A multi-disciplinary view on AI safety research","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/opE6L8jBTTNAyaDbB/a-multi-disciplinary-view-on-ai-safety-research",0,"","interpretability agents"],["Conditioning Predictive Models: Interactions with other approaches","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3ydumADYt9xkaKRTF/conditioning-predictive-models-interactions-with-other",0,"",""],["Dear Anthropic people, please don't release Claude","No drama","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eAEGdZshoYuzYFpMq/dear-anthropic-people-please-don-t-release-claude",0,"","governance"],["[ASoT] Policy Trajectory Visualization","Ulisse Mini","2023","blog","LessWrong","www.lesswrong.com/posts/jMpCXKoCgRp8xmyiN/asot-policy-trajectory-visualization",0,"","interpretability policy"],["[Our World in Data] AI timelines: What do experts in artificial intelligence expect for the future? (Roser, 2023)","Will Aldred","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BsAmChNX9cvwEccny/our-world-in-data-ai-timelines-what-do-experts-in-artificial",0,"","forecasting"],["Conditioning Predictive Models: Making inner alignment as easy as possible","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/qoHwKgLFfPcEuwaba/conditioning-predictive-models-making-inner-alignment-as",0,"",""],["Framing AI strategy","Zach Stein-Perlman","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Svjrk8WRTEyD2HLAZ/framing-ai-strategy",0,"",""],["How evals might (or might not) prevent catastrophic risks from AI","Akash","2023","blog","LessWrong","www.lesswrong.com/posts/SNdijuEn6erTJam3z/how-evals-might-or-might-not-prevent-catastrophic-risks-from",0,"","evals forecasting"],["OpenAI/Microsoft announce \"next generation language model\" integrated into Bing/Edge","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LayCHnetp2jBmmpDx/openai-microsoft-announce-next-generation-language-model",0,"",""],["Review of AI Alignment Progress","PeterMcCluskey","2023","blog","LessWrong","www.lesswrong.com/posts/JqsvYmwzcCKzgE4ZD/review-of-ai-alignment-progress",0,"","interpretability"],["so you think you're not qualified to do technical alignment research?","Tamsin Leake","2023","blog","carado.moe","carado.moe/so-you-think-not-qualified-alignment.html",0,"",""],["so you think you're not qualified to do technical alignment research?","Tamsin Leake","2023","blog","LessWrong","www.lesswrong.com/posts/XkmG8XGf6uhXLmZN7/so-you-think-you-re-not-qualified-to-do-technical-alignment",0,"",""],["tabooing \"AGI\"","Tamsin Leake","2023","blog","carado.moe","carado.moe/tabooing-agi.html",0,"",""],["word report #3","Tamsin Leake","2023","blog","carado.moe","carado.moe/word-report-3.html",0,"",""],["A Toy Model of Universality: Reverse Engineering How Networks Learn Group Operations","Bilal Chughtai and 2 others","2023","paper","arXiv preprint","arxiv.org/abs/2302.03025",0,"","interpretability mechanistic-interpretability"],["Addendum: More Efficient FFNs via Attention","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/ZruH9o8rE7o2NXokv/addendum-more-efficient-ffns-via-attention",0,"","interpretability mechanistic-interpretability"],["Conditioning Predictive Models: The case for competitiveness","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fj8faDDQEfvN2LQcW/conditioning-predictive-models-the-case-for-competitiveness",0,"",""],["Decision Transformer Interpretability","Joseph Bloom and Paul Colognese","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bBuBDJBYHt39Q5zZy/decision-transformer-interpretability",0,"","interpretability"],["Donation recommendations for xrisk + ai safety","vincentweisser","2023","blog","EA Forum","forum.effectivealtruism.org/posts/wEr8XqQvNwf4yP6mx/donation-recommendations-for-xrisk-ai-safety",0,"",""],["Early situational awareness and its implications, a story","Jacob Pfau","2023","blog","LessWrong","www.lesswrong.com/posts/tJzdzGdTGrqFf9ekw/early-situational-awareness-and-its-implications-a-story",0,"","situational-awareness"],["Framing AI strategy","Zach Stein-Perlman","2023","blog","aiimpacts.org","aiimpacts.org/framing-ai-strategy/",0,"",""],["Gradient surfing: the hidden role of regularization","Jesse Hoogland","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fjoy5SxgBmxfy7FNB/gradient-surfing-the-hidden-role-of-regularization",0,"","interpretability"],["Launching The Collective Intelligence Project: Whitepaper and Pilots","jasmine_wang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/suKkvQPxgG6ihvhfP/launching-the-collective-intelligence-project-whitepaper-and",0,"","governance policy"],["SolidGoldMagikarp II: technical details and more recent findings","mwatkins and Jessica Rumbelow","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ya9LzwEbfaAMY8ABo/solidgoldmagikarp-ii-technical-details-and-more-recent",0,"",""],["Are short timelines actually bad?","joshc","2023","blog","LessWrong","www.lesswrong.com/posts/YC6ZjCQPPuKQb49TQ/are-short-timelines-actually-bad",0,"","forecasting"],["Call for submissions: AI Safety Special Session at the Conference on Artificial Life (ALIFE 2023)","Rory Greig","2023","blog","EA Forum","forum.effectivealtruism.org/posts/48Cdimcq7NmznFzkz/call-for-submissions-ai-safety-special-session-at-the",0,"",""],["Evaluations (of new AI Safety researchers) can be noisy","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HACcn8roty9KBAWzZ/evaluations-of-new-ai-safety-researchers-can-be-noisy",0,"","evals"],["Modal Fixpoint Cooperation without Löb's Theorem","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2WpPRrqrFQa6n2x3W/modal-fixpoint-cooperation-without-loeb-s-theorem",0,"","theory"],["Questions about AI that bother me","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/Bhrs7kGkEnDsuCTDa/questions-about-ai-that-bother-me",0,"",""],["SolidGoldMagikarp (plus, prompt generation)","Jessica Rumbelow and mwatkins","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aPeJE8bSo6rAFoLqg/solidgoldmagikarp-plus-prompt-generation",0,"","interpretability robustness"],["A discussion with ChatGPT on value-based models vs. large language models, etc..","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/zmi4oAGMMe92xgSD3/a-discussion-with-chatgpt-on-value-based-models-vs-large",0,"",""],["Attribution Patching: Activation Patching At Industrial Scale","Neel Nanda","2023","report","neelnanda.io","www.neelnanda.io/mechanistic-interpretability/attribution-patching",0,"",""],["AXRP Episode 19 - Mechanistic Interpretability with Neel Nanda","DanielFilan","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/r2yTwkGt3kbQG2mXi/axrp-episode-19-mechanistic-interpretability-with-neel-nanda",0,"","interpretability mechanistic-interpretability"],["Criticism Thread: What things should OpenPhil improve on?","anonymousEA20","2023","blog","EA Forum","forum.effectivealtruism.org/posts/trqswoctpQ92tcY2y/criticism-thread-what-things-should-openphil-improve-on",0,"",""],["Empathy as a natural consequence of learnt reward models","beren","2023","blog","LessWrong","www.lesswrong.com/posts/zaER5ziEprE7aNm6u/empathy-as-a-natural-consequence-of-learnt-reward-models",0,"",""],["Mech Interp Project Advising Call: Memorisation in GPT-2 Small","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kmBeWYKDNtdS3rgQe/mech-interp-project-advising-call-memorisation-in-gpt-2",0,"","interpretability"],["Some miscellaneous thoughts on ChatGPT, stories, and mechanical interpretability","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/t5AfR3LBb6syaxcXn/some-miscellaneous-thoughts-on-chatgpt-stories-and",0,"","interpretability"],["An audio version of the alignment problem from a deep learning perspective by Richard Ngo Et Al","Miguel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/msn9RPhoWRo5AaAFo/an-audio-version-of-the-alignment-problem-from-a-deep",0,"",""],["Assessing China's importance as an AI superpower","JulianHazell","2023","blog","EA Forum","forum.effectivealtruism.org/posts/tBkAg7Cys84eGyew6/assessing-china-s-importance-as-an-ai-superpower",0,"","governance policy"],["ChatGPT: Tantalizing afterthoughts in search of story trajectories [induction heads]","Bill Benzon","2023","blog","LessWrong","www.lesswrong.com/posts/Zcz8otnZuKyExs5g5/chatgpt-tantalizing-afterthoughts-in-search-of-story",0,"","interpretability"],["Focus on the places where you feel shocked everyone’s dropping the ball","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/02/03/focus-on-the-places-where-you-feel-shocked-everyones-dropping-the-ball/",0,"",""],["Google invests $300mn in artificial intelligence start-up Anthropic | FT","𝕮𝖎𝖓𝖊𝖗𝖆","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FnszH6ZGBi9hd8rtv/google-invests-usd300mn-in-artificial-intelligence-start-up",0,"","governance"],["Many AI governance proposals have a tradeoff between usefulness and feasibility","Akash and Carson Ezell","2023","blog","LessWrong","www.lesswrong.com/posts/JHE3ZKCrtvCuxFEMe/many-ai-governance-proposals-have-a-tradeoff-between",0,"","governance"],["What I mean by “alignment is in large part about making cognition aimable at all”","Nate Soares","2023","blog","intelligence.org","intelligence.org/2023/02/02/what-i-mean-by-alignment-is-in-large-part-about-making-cognition-aimable-at-all/",0,"",""],["40,000 reasons to worry about AI safety","Michael Huang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/hN8L9kudPKsdsjKwg/40-000-reasons-to-worry-about-ai-safety",0,"",""],["A Brief Overview of AI Safety/Alignment Orgs, Fields, Researchers, and Resources for ML Researchers","Austin Witte","2023","blog","LessWrong","www.lesswrong.com/posts/2eaLH7zp6pxdQwYSH/a-brief-overview-of-ai-safety-alignment-orgs-fields",0,"",""],["Conditioning Predictive Models: Large language models as predictors","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/XwXmedJAo5m4r29eu/conditioning-predictive-models-large-language-models-as",0,"",""],["Conditioning Predictive Models: Outer alignment via careful conditioning","evhub and 4 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3kkmXfvCv9DmT3kwx/conditioning-predictive-models-outer-alignment-via-careful",0,"",""],["Interviews with 97 AI Researchers: Quantitative Analysis","Maheen Shermohammed and Vael Gates","2023","blog","LessWrong","www.lesswrong.com/posts/Bok5RAPPjuHKvPyv2/interviews-with-97-ai-researchers-quantitative-analysis",0,"",""],["More findings on maximal data dimension","Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/WfdxXhszxFc3BxZ8r/more-findings-on-maximal-data-dimension",0,"","interpretability"],["Normative vs Descriptive Models of Agency","mattmacdermott","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nzRh8yQHi3bx9bLsD/normative-vs-descriptive-models-of-agency-2",0,"","agents theory"],["Predicting researcher interest in AI alignment","Vael Gates","2023","blog","EA Forum","forum.effectivealtruism.org/posts/8pSq73kTJmPrzTfir/predicting-researcher-interest-in-ai-alignment",0,"",""],["Research agenda: Formalizing abstractions of computations","Erik Jenner","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/L8LHBTMvhLDpxDaqv/research-agenda-formalizing-abstractions-of-computations-1",0,"",""],["Retrospective on the AI Safety Field Building Hub","Vael Gates","2023","blog","EA Forum","forum.effectivealtruism.org/posts/n2F2rymJdCcQSYy8y/retrospective-on-the-ai-safety-field-building-hub",0,"",""],["Temporally Layered Architecture for Adaptive, Distributed and Continuous Control","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/468faQvTy7RfG4uxj/temporally-layered-architecture-for-adaptive-distributed-and",0,"","agents"],["You are probably not a good alignment researcher, and other blatant lies","junk heap homotopy","2023","blog","LessWrong","www.lesswrong.com/posts/ZRYqXHdiFrdxLAmue/you-are-probably-not-a-good-alignment-researcher-and-other",0,"","robustness"],["“AI Risk Discussions” website: Exploring interviews from 97 AI Researchers","Vael Gates and 4 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GpnLDSjzNkGB5xvry/ai-risk-discussions-website-exploring-interviews-from-97-ai",0,"",""],["AI Safety Arguments: An Interactive Guide","Lukas Trötzmüller","2023","blog","LessWrong","www.lesswrong.com/posts/9YQby2miskbcKN9FB/ai-safety-arguments-an-interactive-guide",0,"",""],["Eli Lifland on Navigating the AI Alignment Landscape","Ozzie Gooen and elifland","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QeLE22fefLqKfYTW6/eli-lifland-on-navigating-the-ai-alignment-landscape",0,"","forecasting"],["Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 Small","Kevin Ro Wang and 4 others","2023","report","openreview.net","openreview.net/forum?id=NpsVSN6o4ul",0,"","interpretability mechanistic-interpretability"],["Language Models can be Utility-Maximising Agents","Raymond D","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/523mueiug9RapHtWb/language-models-can-be-utility-maximising-agents",0,"","agents"],["More findings on Memorization and double descent","Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/KzwB4ovzrZ8DYWgpw/more-findings-on-memorization-and-double-descent",0,"","interpretability"],["Product safety is a poor model for AI governance","richardkorzekwa","2023","blog","aiimpacts.org","aiimpacts.org/product-safety-is-a-poor-model-for-ai-governance/",0,"","governance"],["Product safety is a poor model for AI governance","Richard Korzekwa","2023","blog","LessWrong","www.lesswrong.com/posts/XTd4xbFSc7NQhAFqh/product-safety-is-a-poor-model-for-ai-governance",0,"","governance"],["The effect of horizon length on scaling laws","Jacob_Hilton","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ouEQLAca3bbiimAFQ/the-effect-of-horizon-length-on-scaling-laws",0,"","scaling-laws"],["Trends in the dollar training cost of machine learning systems","Ben Cottier","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/REBFQF43nwcJgp8Ge/trends-in-the-dollar-training-cost-of-machine-learning-1",0,"","governance forecasting"],["[Linkpost] Human-narrated audio version of \"Is Power-Seeking AI an Existential Risk?\"","Joe_Carlsmith","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yopb28oW9xb8jATjR/linkpost-human-narrated-audio-version-of-is-power-seeking-ai",0,"","power-seeking"],["Apply to HAIST/MAIA’s AI Governance Workshop in DC (Feb 17-20)","Phosphorous and 4 others","2023","blog","LessWrong","www.lesswrong.com/posts/LXgEYLEFbcJnyzSEZ/apply-to-haist-maia-s-ai-governance-workshop-in-dc-feb-17-20",0,"","governance"],["Criticism of the main framework in AI alignment","Michele Campolo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/dTxGyKtshbmWsMWhn/criticism-of-the-main-framework-in-ai-alignment",0,"",""],["How to hedge investment portfolio against AI risk?","Timothy_Liptrot","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cJSuDrGdDW9y7AfBt/how-to-hedge-investment-portfolio-against-ai-risk",0,"",""],["How to use AI speech transcription and analysis to accelerate social science research","AlexanderSaeri","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nxBKxFcfMnEb3Cmys/how-to-use-ai-speech-transcription-and-analysis-to",0,"","policy"],["Inner Misalignment in \"Simulator\" LLMs","Adam Scherlis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/FLMyTjuTiGytE6sP2/inner-misalignment-in-simulator-llms",0,"",""],["Mechanistic Interpretability Quickstart Guide","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jLAvJt8wuSFySN975/mechanistic-interpretability-quickstart-guide",0,"","interpretability mechanistic-interpretability"],["On value in humans, other animals, and AI","Michele Campolo","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/wZyQSMmrFZizmipho/on-value-in-humans-other-animals-and-ai",0,"",""],["Questions about AI that bother me","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/4TcaBNu7EmEukjGoc/questions-about-ai-that-bother-me",0,"",""],["What Are The Biggest Threats To Humanity? (A Happier World video)","Jeroen Willems","2023","blog","EA Forum","forum.effectivealtruism.org/posts/KyNdKTNfJoJccJ7rF/what-are-the-biggest-threats-to-humanity-a-happier-world",0,"",""],["Against Boltzmann mesaoptimizers","porby","2023","blog","LessWrong","www.lesswrong.com/posts/AncrLc5iSc4tmaYBJ/against-boltzmann-mesaoptimizers",0,"",""],["Call for submissions: “(In)human Values and Artificial Agency”, ALIFE 2023","anonymous","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/AGCLZPqtosnd82DmR/call-for-submissions-in-human-values-and-artificial-agency",0,"",""],["Medical Image Registration: The obscure field where Deep Mesaoptimizers are already at the top of the benchmarks. (post + colab notebook)","Hastings","2023","blog","LessWrong","www.lesswrong.com/posts/FgXjuS4R9sRxbzE5w/medical-image-registration-the-obscure-field-where-deep",0,"","benchmarks"],["Model-driven feedback could amplify alignment failures","aogara","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HhBcoRwnyJhQRJnxr/model-driven-feedback-could-amplify-alignment-failures",0,"","rlhf automated-alignment-research"],["Time-stamping: An urgent, neglected AI safety measure","Axel Svensson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QDBntBeBWJ94EQdou/time-stamping-an-urgent-neglected-ai-safety-measure",0,"","interpretability"],["What I mean by \"alignment is in large part about making cognition aimable at all\"","So8res","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/NJYmovr9ZZAyyTBwM/what-i-mean-by-alignment-is-in-large-part-about-making",0,"",""],["Why I hate the \"accident vs. misuse\" AI x-risk dichotomy (quick thoughts on \"structural risk\")","David Scott Krueger (formerly: capybaralet)","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6bpW2kyeKaBtuJuEk/why-i-hate-the-accident-vs-misuse-ai-x-risk-dichotomy-quick",0,"",""],["a guess at my intrinsic values","Tamsin Leake","2023","blog","carado.moe","carado.moe/guess-intrinsic-values.html",0,"",""],["communicating with successful alignment timelines","Tamsin Leake","2023","blog","carado.moe","carado.moe/communicating-successful-alignment.html",0,"","forecasting"],["Compendium of problems with RLHF","Charbel-Raphaël","2023","blog","LessWrong","www.lesswrong.com/posts/d6DvuCKH5bSoT62DB/compendium-of-problems-with-rlhf",0,"","rlhf"],["formal alignment: what it is, and some proposals","Tamsin Leake","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZwEcvG3whyBqBdqSw/formal-alignment-what-it-is-and-some-proposals",0,"",""],["formal alignment: what it is, and some proposals","Tamsin Leake","2023","blog","carado.moe","carado.moe/formal-alignment.html",0,"",""],["Structure, creativity, and novelty","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5pHqQwCDWrvZGp8pX/structure-creativity-and-novelty",0,"",""],["What is the ground reality of countries taking steps to recalibrate AI development towards Alignment first?","anonymous","2023","blog","LessWrong","www.lesswrong.com/posts/whHiGKJBGhgiHi7Ts/what-is-the-ground-reality-of-countries-taking-steps-to",0,"","governance"],["Optimality is the tiger, and annoying the user is its teeth","Christopher King","2023","blog","LessWrong","www.lesswrong.com/posts/N7qE5o3jmoKoe4dHQ/optimality-is-the-tiger-and-annoying-the-user-is-its-teeth",0,"","rlhf"],["Spooky action at a distance in the loss landscape","Jesse Hoogland and Filip Sondej","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2N7eEKDuL5sHQou3N/spooky-action-at-a-distance-in-the-loss-landscape",0,"","interpretability"],["Stop-gradients lead to fixed point predictions","Johannes Treutlein and 3 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/i3v7WeCXyWiYfhihF/stop-gradients-lead-to-fixed-point-predictions",0,"",""],["Assigning Praise and Blame: Decoupling Epistemology and Decision Theory","adamShimi and Gabriel Alfour","2023","blog","LessWrong","www.lesswrong.com/posts/RKzNpJhamYgcLeAEv/assigning-praise-and-blame-decoupling-epistemology-and",0,"","theory"],["Literature review of TAI timelines","Jsevillamol and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/4eAnBaLxvnkydiavw/literature-review-of-tai-timelines",0,"","forecasting"],["The role of Bayesian ML in AI safety - an overview","Marius Hobbhahn","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RJZ7bwoDB6BWgt6St/the-role-of-bayesian-ml-in-ai-safety-an-overview",0,"",""],["to me, it's instrumentality that is alienating","Tamsin Leake","2023","blog","carado.moe","carado.moe/instrumentality-alienating.html",0,"",""],["WaPo: \"Big Tech was moving cautiously on AI. Then came ChatGPT.\"","Julian Bradshaw","2023","blog","LessWrong","www.lesswrong.com/posts/ab8wAd6FJvWAexYug/wapo-big-tech-was-moving-cautiously-on-ai-then-came-chatgpt",0,"","governance"],["\"How to Escape from the Simulation\" - Seeds of Science call for reviewers","rogersbacon1","2023","blog","EA Forum","forum.effectivealtruism.org/posts/TYmufd5pMED6LreJb/how-to-escape-from-the-simulation-seeds-of-science-call-for",0,"",""],["AI Risk Management Framework | NIST","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/biY9kvpStw8QrvD6T/ai-risk-management-framework-or-nist",0,"","governance"],["All AGI Safety questions welcome (especially basic ones) [~monthly thread]","mwatkins and Robert Miles","2023","blog","LessWrong","www.lesswrong.com/posts/SkjTDA98qo83vE6nc/all-agi-safety-questions-welcome-especially-basic-ones-1",0,"",""],["Excerpts from \"Doing EA Better\" on x-risk methodology","BrownHairedEevee","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Wcs96Hao8zN5NgZkS/excerpts-from-doing-ea-better-on-x-risk-methodology",0,"",""],["[RFC] Possible ways to expand on \"Discovering Latent Knowledge in Language Models Without Supervision\".","gekaklam and 3 others","2023","blog","LessWrong","www.lesswrong.com/posts/bFwigCDMC5ishLz7X/rfc-possible-ways-to-expand-on-discovering-latent-knowledge",0,"","interpretability eliciting-latent-knowledge"],["AGI will have learnt utility functions","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/RorXWkriXwErvJtvn/agi-will-have-learnt-utility-functions",0,"",""],["Quick thoughts on \"scalable oversight\" / \"super-human feedback\" research","David Scott Krueger (formerly: capybaralet)","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/4Tx6ALN8erdgRojkk/quick-thoughts-on-scalable-oversight-super-human-feedback",0,"","scalable-oversight rlhf"],["Spreading messages to help with the most important century","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/spreading-messages-to-help-with-the-most-important-century/",0,"",""],["Spreading messages to help with the most important century","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CcJsh4JcxEqYDaSte/spreading-messages-to-help-with-the-most-important-century",0,"",""],["Thoughts on the impact of RLHF research","paulfchristiano","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vwu4kegAEZTBtpT6p/thoughts-on-the-impact-of-rlhf-research",0,"","rlhf"],["Alexander and Yudkowsky on AGI goals","Scott Alexander and Eliezer Yudkowsky","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/rwkkcgSpnAyE8oNo3/alexander-and-yudkowsky-on-agi-goals",0,"",""],["Existential Risk of Misaligned Intelligence Augmentation (Particularly Using High-Bandwidth BCI Implants)","Damian Gorski","2023","blog","EA Forum","forum.effectivealtruism.org/posts/NAdzbiZyJ5rNmBKey/existential-risk-of-misaligned-intelligence-augmentation",0,"",""],["Gradient hacking is extremely difficult","beren","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/w2TAEvME2yAG9MHeq/gradient-hacking-is-extremely-difficult",0,"",""],["How-to Transformer Mechanistic Interpretability—in 50 lines of code or less!","StefanHex","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hnzHrdqn3nrjveayv/how-to-transformer-mechanistic-interpretability-in-50-lines",0,"","interpretability mechanistic-interpretability"],["Inverse Scaling Prize: Second Round Winners","Ian McKenzie and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DARiTSTx5xDLQGrrz/inverse-scaling-prize-second-round-winners",0,"","scaling-laws"],["Large Language Models as Fiduciaries to Humans","johnjnay","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cWeioTmbs73iZjs25/large-language-models-as-fiduciaries-to-humans",0,"","governance policy"],["Parameter Scaling Comes for RL, Maybe","1a3orn","2023","blog","LessWrong","www.lesswrong.com/posts/4xGAmZ9GTGAkszHoH/parameter-scaling-comes-for-rl-maybe",0,"","scaling-laws"],["Some of my disagreements with List of Lethalities","TurnTrout","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/kpFxkXBbpF5pWDRrc/some-of-my-disagreements-with-list-of-lethalities",0,"",""],["Thoughts on hardware / compute requirements for AGI","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/LY7rovMiJ4FhHxmH5/thoughts-on-hardware-compute-requirements-for-agi",0,"","governance forecasting"],["Update to Samotsvety AGI timelines","Misha_Yagudin and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ByBBqwRXWqX5m9erL/update-to-samotsvety-agi-timelines",0,"","forecasting"],["Why people want to work on AI safety (but don’t)","Emily Grundy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/QWuKM5fsbry8Jp2x5/why-people-want-to-work-on-ai-safety-but-don-t",0,"",""],["“Endgame safety” for AGI","Steven Byrnes","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/MCWGCyz2mjtRoWiyP/endgame-safety-for-agi",0,"","deception"],["AI safety milestones?","Zach Stein-Perlman","2023","blog","LessWrong","www.lesswrong.com/posts/bv2diKZv3FBGHnXHe/ai-safety-milestones",0,"","governance"],["Has private AGI research made independent safety research ineffective already? What should we do about this?","Roman Leventov","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ZExF3Z7WZpdBZZZEy/has-private-agi-research-made-independent-safety-research",0,"","governance policy"],["My highly personal skepticism braindump on existential risk from artificial intelligence.","NunoSempere","2023","blog","EA Forum","forum.effectivealtruism.org/posts/L6ZmggEJw8ri4KB8X/my-highly-personal-skepticism-braindump-on-existential-risk",0,"","forecasting"],["There should be a public adversarial collaboration on AI x-risk","pradyuprasad","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nRXugEFFDz7MtGKz9/there-should-be-a-public-adversarial-collaboration-on-ai-x",0,"",""],["What a compute-centric framework says about AI takeoff speeds","Tom Davidson","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gc9FGtdXhK9sCSEYu/what-a-compute-centric-framework-says-about-ai-takeoff",0,"","forecasting"],["What a compute-centric framework says about AI takeoff speeds","Tom_Davidson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/3vDarp6adLPBTux5g/what-a-compute-centric-framework-says-about-ai-takeoff",0,"","forecasting"],["Emotional attachment to AIs opens doors to problems","Igor Ivanov","2023","blog","LessWrong","www.lesswrong.com/posts/w9oACum6FW7HdGHST/emotional-attachment-to-ais-opens-doors-to-problems",0,"","governance"],["Gemini modeling","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HJ4EHPG5qPbbbk5nK/gemini-modeling",0,"",""],["Large language models learn to represent the world","gjm","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Neh76ueECviJ6p75o/large-language-models-learn-to-represent-the-world",0,"","interpretability"],["NYT: Google will ‘recalibrate’ the risk of releasing AI due to competition with OpenAI","Michael Huang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Nm9ahJzKsDGFfF66b/nyt-google-will-recalibrate-the-risk-of-releasing-ai-due-to",0,"",""],["AI Safety \"Textbook\". Test chapter. Orthogonality Thesis, Goodhart Law and Instrumental Convergency","Tapatakt and LacrimalBird","2023","blog","LessWrong","www.lesswrong.com/posts/j6mcesYTasNG2zKWd/ai-safety-textbook-test-chapter-orthogonality-thesis",0,"","goodharts-law"],["Announcing aisafety.training","JJ Hepburn","2023","blog","LessWrong","www.lesswrong.com/posts/MKvtmNGCtwNqc44qm/announcing-aisafety-training",0,"",""],["We Ran an Alignment Workshop","aiden ament","2023","blog","EA Forum","forum.effectivealtruism.org/posts/SkkAo8W4rg5kGrkTc/we-ran-an-alignment-workshop",0,"",""],["[TIME magazine] DeepMind’s CEO Helped Take AI Mainstream. Now He’s Urging Caution (Perrigo, 2023)","Will Aldred","2023","blog","EA Forum","forum.effectivealtruism.org/posts/GDkrPrP2m6TQqdSGF/time-magazine-deepmind-s-ceo-helped-take-ai-mainstream-now",0,"",""],["Critique of some recent philosophy of LLMs’ minds","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ejEgaYSaefCevapPa/critique-of-some-recent-philosophy-of-llms-minds",0,"",""],["Shard theory alignment has important, often-overlooked free parameters.","Charlie Steiner","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/uz2mdPtdBnaXpXPmT/shard-theory-alignment-has-important-often-overlooked-free",0,"",""],["Transcript of Sam Altman's interview touching on AI safety","Andy_McKenzie","2023","blog","LessWrong","www.lesswrong.com/posts/PTzsEQXkCfig9A6AS/transcript-of-sam-altman-s-interview-touching-on-ai-safety",0,"",""],["What’s going on with ‘crunch time’?","rosehadshar","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7CdtdieiijWXWhiZB/what-s-going-on-with-crunch-time",0,"",""],["\"Heretical Thoughts on AI\" by Eli Dourado","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/nWCokT9xbrY4p98co/heretical-thoughts-on-ai-by-eli-dourado",0,"","forecasting"],["200 COP in MI: Studying Learned Features in Language Models","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qup9gorqpd9qKAEav/200-cop-in-mi-studying-learned-features-in-language-models",0,"","interpretability"],["6-paragraph AI risk intro for MAISI","JakubK","2023","blog","LessWrong","www.lesswrong.com/posts/coe8zoG7t2DZrC4xY/6-paragraph-ai-risk-intro-for-maisi",0,"",""],["AGI safety field building projects I’d like to see","Severin T. Seehrich","2023","blog","LessWrong","www.lesswrong.com/posts/QRST9ctX5Cu2dM2Sb/agi-safety-field-building-projects-i-d-like-to-see",0,"",""],["Announcing Cavendish Labs","dyusha and Derik K","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xBeqaWEJfWZv8ALWn/announcing-cavendish-labs",0,"",""],["Announcing Cavendish Labs","derikk and agg","2023","blog","LessWrong","www.lesswrong.com/posts/JiGwL5nDMTwehP62N/announcing-cavendish-labs",0,"",""],["List of technical AI safety exercises and projects","JakubK","2023","blog","LessWrong","www.lesswrong.com/posts/rmCqibBhytQizcief/list-of-technical-ai-safety-exercises-and-projects",0,"",""],["nostalgia: a value pointing home","Tamsin Leake","2023","blog","carado.moe","carado.moe/nostalgia.html",0,"",""],["Thoughts on refusing harmful requests to large language models","William_S","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Dan6iKFruioYmZafw/thoughts-on-refusing-harmful-requests-to-large-language",0,"",""],["Any Philosophy PhD recommendations for students interested in Alignment Efforts?","rickyhuang.hexuan","2023","blog","EA Forum","forum.effectivealtruism.org/posts/eXwZEjGrx6JJgtEPG/any-philosophy-phd-recommendations-for-students-interested",0,"",""],["Approfondimenti sui rischi dell’IA (materiali in inglese)","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/gEExWiqRnvkGbBpPc/approfondimenti-sui-rischi-dell-ia-materiali-in-inglese",0,"",""],["Emerging Paradigms: The Case of Artificial Intelligence Safety","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pC9RJdmP3rnhuHpCm/emerging-paradigms-the-case-of-artificial-intelligence",0,"",""],["Gradient Filtering","Jozdien and janus","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2sTTEkzvscWCPBQAk/gradient-filtering",0,"",""],["Help me to understand AI alignment!","britomart","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ph2ETEDuWhzmHxvsi/help-me-to-understand-ai-alignment",0,"",""],["Neural networks generalize because of this one weird trick","Jesse Hoogland","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fovfuFdpuEwQzJu2w/neural-networks-generalize-because-of-this-one-weird-trick",0,"","interpretability"],["Vitalik on science, his philanthropy and effective altruism.","vincentweisser","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LBhd3Pqw2ehgykQem/vitalik-on-science-his-philanthropy-and-effective-altruism",0,"",""],["AGISF adaptation for in-person groups","Sam Marks and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/foJhEZzG5sx9cQJZq/agisf-adaptation-for-in-person-groups",0,"",""],["Collin Burns on Alignment Research And Discovering Latent Knowledge Without Supervision","Michaël Trazzi","2023","blog","LessWrong","www.lesswrong.com/posts/rWnQeqfCg3M2DXdeb/collin-burns-on-alignment-research-and-discovering-latent",0,"","eliciting-latent-knowledge"],["How many people are working (directly) on reducing existential risk from AI?","Benjamin Hilton and 80000_Hours","2023","blog","EA Forum","forum.effectivealtruism.org/posts/rZoRGxJzipcQoaPST/how-many-people-are-working-directly-on-reducing-existential",0,"","forecasting"],["Il panorama della governance lungoterminista delle intelligenze artificiali","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/BkcnuZyKcDZBpSrQS/il-panorama-della-governance-lungoterminista-delle",0,"","governance"],["Le Tempistiche delle IA: il dibattito e il punto di vista degli “esperti”","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/pvJxxpza2ZRa5y9e2/le-tempistiche-delle-ia-il-dibattito-e-il-punto-di-vista",0,"","forecasting"],["Lessons learned and review of the AI Safety Nudge Competition","Marc Carauleanu and Chris Leong","2023","blog","EA Forum","forum.effectivealtruism.org/posts/AGhrkv7Ha6giWon7Z/lessons-learned-and-review-of-the-ai-safety-nudge",0,"",""],["Löbian emotional processing of emergent cooperation: an example","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/GwihvYHE3u6s3LbeM/loebian-emotional-processing-of-emergent-cooperation-an",0,"",""],["L’importanza delle IA come possibile minaccia per l’umanità","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/xqBaBjXYy5yHbpXou/l-importanza-delle-ia-come-possibile-minaccia-per-l-umanita",0,"",""],["Perché il deep learning moderno potrebbe rendere difficile l’allineamento delle IA","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LB4b4idcMCWg4eJYA/perche-il-deep-learning-moderno-potrebbe-rendere-difficile-l",0,"",""],["Preparing for AI-assisted alignment research: we need data!","CBiddulph","2023","blog","EA Forum","forum.effectivealtruism.org/posts/WZf6KpmajZXs596JG/preparing-for-ai-assisted-alignment-research-we-need-data",0,"",""],["Prevenire una catastrofe legata all'intelligenza artificiale","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/sxcex5KomcHgojzhc/prevenire-una-catastrofe-legata-all-intelligenza-artificiale",0,"",""],["Ricerca sulla sicurezza delle IA: panoramica delle carriere","EA Italy","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Yoxmaj8QKpQun9mDz/ricerca-sulla-sicurezza-delle-ia-panoramica-delle-carriere",0,"",""],["Should AI writers be prohibited in education?","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/TjKmAu6xjezwNhTBK/should-ai-writers-be-prohibited-in-education",0,"","governance"],["What can thought-experiments do?","Cleo Nardo","2023","blog","LessWrong","www.lesswrong.com/posts/F6jAKPDMPdyAEuCPQ/what-can-thought-experiments-do",0,"","theory"],["Aligning the Aligners: Ensuring Aligned AI acts for the common good of all mankind","timunderwood","2023","blog","EA Forum","forum.effectivealtruism.org/posts/nJz6vaqK7xBDXpMwk/aligning-the-aligners-ensuring-aligned-ai-acts-for-the",0,"","governance robustness"],["Can GPT-3 produce new ideas? Partially automating Robin Hanson and others","NunoSempere","2023","blog","EA Forum","forum.effectivealtruism.org/posts/63pYakESGrQpfNw25/can-gpt-3-produce-new-ideas-partially-automating-robin",0,"",""],["Consequentialists: One-Way Pattern Traps","David Udell","2023","blog","LessWrong","www.lesswrong.com/posts/eD34hTMp8uv3ifSjg/consequentialists-one-way-pattern-traps",0,"","theory"],["EA relevant Foresight Institute Workshops in 2023: WBE & AI safety, Cryptography & AI safety, XHope, Space, and Atomically Precise Manufacturing","elteerkers","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CNxteiKdRk9Hez3pv/ea-relevant-foresight-institute-workshops-in-2023-wbe-and-ai",0,"",""],["Experiment Idea: RL Agents Evading Learned Shutdownability","Leon Lang","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y59AYj5keDYHf29LK/experiment-idea-rl-agents-evading-learned-shutdownability",0,"","agents"],["How we could stumble into AI catastrophe","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yjm5CW9JdwBTFZB2B/how-we-could-stumble-into-ai-catastrophe",0,"","forecasting"],["Import AI - coming soon to Substack","Jack Clark","2023","blog","importai.substack.com","importai.substack.com/p/coming-soon",0,"",""],["Reflections on Trusting Trust & AI","Itay Yona","2023","blog","LessWrong","www.lesswrong.com/posts/BMnhDjJrix5BXE7yr/reflections-on-trusting-trust-and-ai",0,"","interpretability"],["Should AI writers be prohibited in education?","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yNitwYkHP6DtkkSrG/should-ai-writers-be-prohibited-in-education",0,"","evals governance policy"],["Showing versus doing: Teaching by demonstration","M. K. Ho and 4 others","2023","report","par.nsf.gov","par.nsf.gov/biblio/10082788-showing-versus-doing-teaching-demonstration",0,"",""],["Deceptive failures short of full catastrophe.","Alex Lawsen","2023","blog","LessWrong","www.lesswrong.com/posts/iNaB6GA6Seti3biTJ/deceptive-failures-short-of-full-catastrophe",0,"","alignment-faking deception"],["Non-directed conceptual founding","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JFoQ3echrH2pfjKuP/non-directed-conceptual-founding",0,"",""],["Speculation on Path-Dependance in Large Language Models.","NickyP","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/pt8Sf2kvRZ8BBW5b5/speculation-on-path-dependance-in-large-language-models",0,"",""],["Underspecification of Oracle AI","Rubi J. Hudson and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/aBRS3x4sPSJ9G6xkj/underspecification-of-oracle-ai",0,"",""],["Concrete Reasons for Hope about AI","Zac Hatfield-Dodds","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BfN88BfZQ4XGeZkda/concrete-reasons-for-hope-about-ai",0,"",""],["World-Model Interpretability Is All We Need","Thane Ruthenis","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/HaHcsrDSZ3ZC2b4fK/world-model-interpretability-is-all-we-need",0,"","interpretability"],["[ASoT] Simulators show us behavioural properties by default","Jozdien","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/mC2omdN4ekcsNkCmp/asot-simulators-show-us-behavioural-properties-by-default-1",0,"","eliciting-latent-knowledge chain-of-thought-faithfulness"],["AGISF adaptation for in-person groups","Sam Marks and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/jsy8t64Jp5xcuFrXX/agisf-adaptation-for-in-person-groups",0,"",""],["Beware safety-washing","Lizka","2023","blog","EA Forum","forum.effectivealtruism.org/posts/f2qojPr8NaMPo2KJC/beware-safety-washing",0,"",""],["Concerns about AI safety career change","mmKALLL","2023","blog","EA Forum","forum.effectivealtruism.org/posts/kDyG6p6FqwJ4ioQt4/concerns-about-ai-safety-career-change",0,"",""],["Disentangling Shard Theory into Atomic Claims","Leon Lang","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/L4e7CqqpDxea2x4Gg/disentangling-shard-theory-into-atomic-claims",0,"",""],["How does GPT-3 spend its 175B parameters?","Robert_AIZI","2023","blog","LessWrong","www.lesswrong.com/posts/3duR8CrvcHywrnhLo/how-does-gpt-3-spend-its-175b-parameters",0,"","interpretability"],["How we could stumble into AI catastrophe","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/how-we-could-stumble-into-ai-catastrophe/",0,"",""],["Some Arguments Against Strong Scaling","Joar Skalse","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/DvCLEkr9pXLnWikB8/some-arguments-against-strong-scaling",0,"","forecasting scaling-laws"],["The AI Control Problem in a wider intellectual context","philosophybear","2023","blog","LessWrong","www.lesswrong.com/posts/Afs6FtptMSWcAetxR/the-ai-control-problem-in-a-wider-intellectual-context",0,"","interpretability ai-control"],["Tracr: Compiled Transformers as a Laboratory for Interpretability | DeepMind","DragonGod","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/2J6fFHQZkWxFcjL6c/tracr-compiled-transformers-as-a-laboratory-for-1",0,"","interpretability"],["Tracr: Compiled Transformers as a Laboratory for Interpretability | DeepMind","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/2J6fFHQZkWxFcjL6c/tracr-compiled-transformers-as-a-laboratory-for-1",0,"","interpretability"],["[Linkpost] Scaling Laws for Generative Mixed-Modal Language Models","Amal","2023","blog","LessWrong","www.lesswrong.com/posts/6w9uTPdJk52Nyknvm/linkpost-scaling-laws-for-generative-mixed-modal-language",0,"","scaling-laws"],["Alignment is not enough","Alan Chan","2023","blog","LessWrong","www.lesswrong.com/posts/nEzFkaQKPjNnmqfEm/alignment-is-not-enough",0,"","governance"],["Announcing the 2023 PIBBSS Summer Research Fellowship","Nora_Ammann and DusanDNesic","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cJWipZCfAbg9tFX2P/announcing-the-2023-pibbss-summer-research-fellowship",0,"",""],["Announcing the 2023 PIBBSS Summer Research Fellowship","Dušan D. Nešić (Dushan) and nora","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mqBLFdNzkxfbfcaoX/announcing-the-2023-pibbss-summer-research-fellowship",0,"",""],["Categorical-measure-theoretic approach to optimal policies tending to seek power","jacek","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9dJgvGh4wNyPbQfrt/categorical-measure-theoretic-approach-to-optimal-policies",0,"","power-seeking"],["ChatGPT struggles to respond to the real world","Alex Flint","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/H4bAKBwnFktvoLtp4/chatgpt-struggles-to-respond-to-the-real-world",0,"",""],["ea.domains - Domains Free to a Good Home","plex and Alignment Ecosystem Development","2023","blog","EA Forum","forum.effectivealtruism.org/posts/LHN8mfi9Dc7bKD6Gu/ea-domains-domains-free-to-a-good-home",0,"","robustness"],["How it feels to have your mind hacked by an AI","blaked","2023","blog","LessWrong","www.lesswrong.com/posts/9kQFure4hdDmRBNdH/how-it-feels-to-have-your-mind-hacked-by-an-ai",0,"",""],["Microsoft Plans to Invest $10B in OpenAI; $3B Invested to Date | Fortune","DragonGod","2023","blog","LessWrong","www.lesswrong.com/posts/X7QbzyKWqnLmeZCnJ/microsoft-plans-to-invest-usd10b-in-openai-usd3b-invested-to",0,"","forecasting"],["ML Summer Bootcamp Reflection: Aalto EA Finland","Aayush Kucheria","2023","blog","EA Forum","forum.effectivealtruism.org/posts/j9nLvT5ej8mKc4fhi/ml-summer-bootcamp-reflection-aalto-ea-finland",0,"",""],["Reward is not Necessary: How to Create a Compositional Self-Preserving Agent for Life-Long Learning","Roman Leventov","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/df4Jjg9cmJ7R2bkzR/reward-is-not-necessary-how-to-create-a-compositional-self-1",0,"","agents theory"],["The Alignment Problems","Martín Soto","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/cHd6xLX6qeNYQACat/the-alignment-problems-1",0,"",""],["Victoria Krakovna on AGI Ruin, The Sharp Left Turn and Paradigms of AI Alignment","Michaël Trazzi","2023","blog","LessWrong","www.lesswrong.com/posts/5SqSZazHjrzhvxmCE/victoria-krakovna-on-agi-ruin-the-sharp-left-turn-and",0,"",""],["Forecasting potential misuses of language models for disinformation campaigns and how to reduce risk","Josh A. Goldstein and 5 others","2023","blog","openai.com","openai.com/research/forecasting-misuse",0,"","forecasting"],["200 COP in MI: Interpreting Reinforcement Learning","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/eqvvDM25MXLGqumnf/200-cop-in-mi-interpreting-reinforcement-learning",0,"","interpretability"],["Against using stock prices to forecast AI timelines","basil.halperin and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/izJxJwgteyDrKyyXe/against-using-stock-prices-to-forecast-ai-timelines",0,"","forecasting"],["Against using stock prices to forecast AI timelines","basil.halperin and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/yFkNYyspBBqfSeBx9/against-using-stock-prices-to-forecast-ai-timelines",0,"","forecasting"],["AGI and the EMH: markets are not expecting aligned or unaligned AI in the next 30 years","basil.halperin and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/ngpC5PFAgxHJMhicM/agi-and-the-emh-markets-are-not-expecting-aligned-or-1",0,"","forecasting"],["Review AI Alignment posts to help figure out how to make a proper AI Alignment review","habryka and Raemon","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qprn8tMeZGLBfobu7/review-ai-alignment-posts-to-help-figure-out-how-to-make-a",0,"",""],["The Alignment Problem from a Deep Learning Perspective (major rewrite)","SoerenMind and 2 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/5GxLiJJEzvqmTNyCK/the-alignment-problem-from-a-deep-learning-perspective-major",0,"",""],["We don’t trade with ants","Katja Grace","2023","blog","aiimpacts.org","aiimpacts.org/we-dont-trade-with-ants/",0,"",""],["What AI Take-Over Movies or Books Will Scare Me Into Taking AI Seriously?","Jordan Arel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/C5GxzWrJRrPibia5z/what-ai-take-over-movies-or-books-will-scare-me-into-taking",0,"",""],["[MLSN #7]: an example of an emergent internal optimizer","joshc and Dan H","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/BsbmsrpboN5jbESwD/mlsn-7-an-example-of-an-emergent-internal-optimizer",0,"",""],["Big list of AI safety videos","JakubK","2023","blog","LessWrong","www.lesswrong.com/posts/H2BPqnvv7YyjiEHam/big-list-of-ai-safety-videos",0,"",""],["Is anyone else also getting more worried about hard takeoff AGI scenarios?","JonCefalu","2023","blog","EA Forum","forum.effectivealtruism.org/posts/2bQrhgkK2DxLtNGbj/is-anyone-else-also-getting-more-worried-about-hard-takeoff",0,"","forecasting"],["ML Safety Newsletter #7","Dan Hendrycks","2023","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-7",0,"",""],["Nearcast-based “deployment problem” analysis (Karnofsky, 2022)","Will Aldred","2023","blog","EA Forum","forum.effectivealtruism.org/posts/CKNJ9Lxru34JevCyi/nearcast-based-deployment-problem-analysis-karnofsky-2022",0,"",""],["Trying to isolate objectives: approaches toward high-level interpretability","Jozdien","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/sDKi2pQ3fnTSbR7H8/trying-to-isolate-objectives-approaches-toward-high-level",0,"","interpretability"],["Wentworth and Larsen on buying time","Akash and 2 others","2023","blog","LessWrong","www.lesswrong.com/posts/JdGuqg7ifRwPiirCe/wentworth-and-larsen-on-buying-time",0,"","governance"],["You're Not One \"You\" - How Decision Theories Are Talking Past Each Other","keith_wynroe","2023","blog","LessWrong","www.lesswrong.com/posts/4gaGgGs5mEara9rea/you-re-not-one-you-how-decision-theories-are-talking-past",0,"","theory"],["200 COP in MI: Image Model Interpretability","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/caMoe6yNfXcaCG2u3/200-cop-in-mi-image-model-interpretability",0,"","interpretability"],["Is this community over-emphasizing AI alignment?","Lixiang","2023","blog","EA Forum","forum.effectivealtruism.org/posts/jSJk9BPTCuHo7Acv7/is-this-community-over-emphasizing-ai-alignment",0,"",""],["Learning as much Deep Learning math as I could in 24 hours","Phosphorous","2023","blog","EA Forum","forum.effectivealtruism.org/posts/Rnga2XRJzeYypyXDt/learning-as-much-deep-learning-math-as-i-could-in-24-hours",0,"",""],["Research ideas (AI Interpretability & Neurosciences) for a 2-months project","flux","2023","blog","LessWrong","www.lesswrong.com/posts/KDmo23saeq5GegTbA/research-ideas-ai-interpretability-and-neurosciences-for-a-2",0,"","interpretability"],["Simulacra are Things","janus","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/3BDqZMNSJDBg2oyvW/simulacra-are-things",0,"",""],["David Krueger on AI Alignment in Academia and Coordination","Michaël Trazzi","2023","blog","EA Forum","forum.effectivealtruism.org/posts/EJ5a2ApokQqGB98P8/david-krueger-on-ai-alignment-in-academia-and-coordination",0,"",""],["How to create curriculum for self-study towards AI alignment work?","OIUJHKDFS","2023","blog","EA Forum","forum.effectivealtruism.org/posts/7KL8CitpBmnzZgKHY/how-to-create-curriculum-for-self-study-towards-ai-alignment",0,"",""],["Looking for Spanish AI Alignment Researchers","Antb","2023","blog","LessWrong","www.lesswrong.com/posts/oxT9WJNzTG9ESjcPK/looking-for-spanish-ai-alignment-researchers",0,"",""],["Protectionism will Slow the Deployment of AI","bgold","2023","blog","LessWrong","www.lesswrong.com/posts/apdXGcQJNuCSrgg4x/protectionism-will-slow-the-deployment-of-ai",0,"","governance"],["200 COP in MI: Techniques, Tooling and Automation","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/btasQF7wiCYPsr5qw/200-cop-in-mi-techniques-tooling-and-automation",0,"","interpretability mechanistic-interpretability"],["2022-23 New Year review","Victoria Krakovna","2023","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2023/01/06/2022-23-new-year-review/",0,"",""],["AI Safety Camp, Virtual Edition 2023","Linda Linsefors","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9AXSrp5MAThZZEfTc/ai-safety-camp-virtual-edition-2023",0,"",""],["AI Safety Camp: Machine Learning for Scientific Discovery","Eleni Angelou","2023","blog","LessWrong","www.lesswrong.com/posts/oQh89BfH4aRthPiwY/ai-safety-camp-machine-learning-for-scientific-discovery-1",0,"",""],["AI security might be helpful for AI alignment","Igor Ivanov","2023","blog","LessWrong","www.lesswrong.com/posts/tDmkHz9ZLdHn2kzp9/ai-security-might-be-helpful-for-ai-alignment",0,"","governance"],["Categorizing failures as “outer” or “inner” misalignment is often confused","Rohin Shah","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JKwrDwsaRiSxTv9ur/categorizing-failures-as-outer-or-inner-misalignment-is",0,"",""],["Definitions of “objective” should be Probable and Predictive","Rohin Shah","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/ASoGszmr9C5MPLtpC/definitions-of-objective-should-be-probable-and-predictive",0,"",""],["Machine Learning for Scientific Discovery - AI Safety Camp","Eleni_A","2023","blog","EA Forum","forum.effectivealtruism.org/posts/9qWknhfxrgtMoD4J9/machine-learning-for-scientific-discovery-ai-safety-camp",0,"",""],["Metaculus Year in Review: 2022","christian","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cbtoajkfeXqJAzhRi/metaculus-year-in-review-2022",0,"","policy forecasting"],["Transformative AI issues (not just misalignment): an overview","Holden Karnofsky","2023","blog","EA Forum","forum.effectivealtruism.org/posts/mPkFheB4EM6pmEC7y/transformative-ai-issues-not-just-misalignment-an-overview",0,"",""],["Illusion of truth effect and Ambiguity effect: Bias in Evaluating AGI X-Risks","Remmelt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/ExtCWHofqmBwDqfcb/illusion-of-truth-effect-and-ambiguity-effect-bias-in",0,"","evals"],["Paper: Superposition, Memorization, and Double Descent (Anthropic)","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Ks6p33LQyfFkNtYE/paper-superposition-memorization-and-double-descent",0,"","interpretability mechanistic-interpretability"],["Skill up in ML for AI safety with the Intro to ML Safety course (Spring 2023)","james and Oliver Z","2023","blog","EA Forum","forum.effectivealtruism.org/posts/aWr4rMf7ZhoCAtoMc/skill-up-in-ml-for-ai-safety-with-the-intro-to-ml-safety",0,"",""],["Superposition, Memorization, and Double Descent","Tom Henighan and 7 others","2023","blog","transformer-circuits.pub","transformer-circuits.pub/2023/toy-double-descent/index.html",0,"","mechanistic-interpretability"],["Transformative AI issues (not just misalignment): an overview","Holden Karnofsky","2023","blog","cold-takes.com","www.cold-takes.com/transformative-ai-issues-not-just-misalignment-an-overview/",0,"",""],["When you plan according to your AI timelines, should you put more weight on the median future, or the median future | eventual AI alignment success? ⚖️","Jeffrey Ladish","2023","blog","EA Forum","forum.effectivealtruism.org/posts/cPuTnDowko79KAcn3/when-you-plan-according-to-your-ai-timelines-should-you-put",0,"","forecasting"],["Why I'm joining Anthropic","evhub","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/7jn5aDadcMH6sFeJe/why-i-m-joining-anthropic",0,"",""],["200 COP in MI: Analysing Training Dynamics","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/hHaXzJQi6SKkeXzbg/200-cop-in-mi-analysing-training-dynamics",0,"","interpretability mechanistic-interpretability"],["2022 was the year AGI arrived (Just don't call it that)","Logan Zoellner","2023","blog","LessWrong","www.lesswrong.com/posts/HguqQSY8mR7NxGopc/2022-was-the-year-agi-arrived-just-don-t-call-it-that",0,"","forecasting"],["Announcing Insights for Impact","Christian Pearson","2023","blog","EA Forum","forum.effectivealtruism.org/posts/iuBoizzA5c5KfWysc/announcing-insights-for-impact",0,"","policy"],["Basic Facts about Language Model Internals","beren and Eric Winsor","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/PDLfpRwSynu73mxGw/basic-facts-about-language-model-internals-1",0,"","interpretability"],["Causal representation learning as a technique to prevent goal misgeneralization","PabloAMC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/vzLNnc59LyzqDCyJC/causal-representation-learning-as-a-technique-to-prevent",0,"","theory"],["ChatGPT understands, but largely does not generate Spanglish (and other code-mixed) text","Milan Weibel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5f9Xtmy8Q559eppqJ/chatgpt-understands-but-largely-does-not-generate-spanglish",0,"",""],["Contra Common Knowledge","abramdemski","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/bG8u3HDZ5AQDJhtTk/contra-common-knowledge",0,"",""],["Large Language Models as Corporate Lobbyists, and Implications for Societal-AI Alignment","johnjnay","2023","blog","EA Forum","forum.effectivealtruism.org/posts/yFQREgJtKib7zGM9w/large-language-models-as-corporate-lobbyists-and",0,"","governance policy"],["List of links for getting into AI safety","zef","2023","blog","LessWrong","www.lesswrong.com/posts/FkDuWGtiCTshovoTN/list-of-links-for-getting-into-ai-safety",0,"",""],["Normalcy bias and Base rate neglect: Bias in Evaluating AGI X-Risks","Remmelt","2023","blog","EA Forum","forum.effectivealtruism.org/posts/daLssjprpqfAsRWW8/normalcy-bias-and-base-rate-neglect-bias-in-evaluating-agi-x",0,"","evals"],["200 COP in MI: Exploring Polysemanticity and Superposition","Neel Nanda","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/o6ptPu7arZrqRCxyz/200-cop-in-mi-exploring-polysemanticity-and-superposition",0,"","interpretability mechanistic-interpretability"],["Holden Karnofsky Interview about Most Important Century & Transformative AI","Dwarkesh Patel","2023","blog","EA Forum","forum.effectivealtruism.org/posts/MYxerzJrCzHErKWX6/holden-karnofsky-interview-about-most-important-century-and",0,"",""],["How have shorter AI timelines been affecting you, and how have you been responding to them?","Liav.Koren","2023","blog","EA Forum","forum.effectivealtruism.org/posts/FGiKbbTKezqj9bvbm/how-have-shorter-ai-timelines-been-affecting-you-and-how",0,"","forecasting"],["I have thousands of copies of HPMOR in Russian. How to use them with the most impact?","Mikhail Samin","2023","blog","LessWrong","www.lesswrong.com/posts/6qkBM73ea5dmJm5nY/i-have-thousands-of-copies-of-hpmor-in-russian-how-to-use",0,"",""],["Is recursive self-alignment possible?","No77e","2023","blog","LessWrong","www.lesswrong.com/posts/s9aB6fLiAmd2d8GRK/is-recursive-self-alignment-possible",0,"","forecasting"],["Touch reality as soon as possible (when doing machine learning research)","LawrenceC","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/fqryrxnvpSr5w2dDJ/touch-reality-as-soon-as-possible-when-doing-machine",0,"",""],["Whisper's Wild Implications","Ollie J","2023","blog","LessWrong","www.lesswrong.com/posts/KbRxdBCcJqwtbiPzm/whisper-s-wild-implications-1",0,"","scaling-laws"],["[Simulators seminar sequence] #1 Background & shared assumptions","Jan and 9 others","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/nmMorGE4MS4txzr8q/simulators-seminar-sequence-1-background-and-shared",0,"",""],["AI Safety Doesn't Have to be Weird","Mica White","2023","blog","EA Forum","forum.effectivealtruism.org/posts/vDvcRQ8yeh6XXoHgx/ai-safety-doesn-t-have-to-be-weird",0,"",""],["Alignment, Anger, and Love: Preparing for the Emergence of Superintelligent AI","tavurth","2023","blog","LessWrong","www.lesswrong.com/posts/pQFpkwiQNjQzjGzCn/alignment-anger-and-love-preparing-for-the-emergence-of",0,"",""],["Large language models can provide \"normative assumptions\" for learning human preferences","Stuart_Armstrong","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/JSkqkgYcyYt8oHsFi/large-language-models-can-provide-normative-assumptions-for",0,"",""],["My first year in AI alignment","Alex_Altair","2023","blog","LessWrong","www.lesswrong.com/posts/rTJrqtDLxAPxiW3sk/my-first-year-in-ai-alignment",0,"",""],["On the Importance of Open Sourcing Reward Models","elandgre","2023","blog","LessWrong","www.lesswrong.com/posts/nTy48zvBPPttoLhdJ/on-the-importance-of-open-sourcing-reward-models",0,"","rlhf interpretability"],["Results from the AI testing hackathon","Esben Kran and 2 others","2023","blog","EA Forum","forum.effectivealtruism.org/posts/5h8bNTFHkrNNzrrJf/results-from-the-ai-testing-hackathon",0,"",""],["Soft optimization makes the value target bigger","Jeremy Gillen","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/9fL22eBJMtyCLvL7j/soft-optimization-makes-the-value-target-bigger",0,"","goodharts-law"],["A Löbian argument pattern for implicit reasoning in natural language: Löbian party invitations","Andrew_Critch","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/QsZ3ycfRYs2ps5sNA/a-loebian-argument-pattern-for-implicit-reasoning-in-natural",0,"",""],["Summary of 80k's AI problem profile","JakubK","2023","blog","LessWrong","www.lesswrong.com/posts/hjL4KnPKtoa3JGtPy/summary-of-80k-s-ai-problem-profile",0,"",""],["The Thingness of Things","TsviBT","2023","blog","AI Alignment Forum","www.alignmentforum.org/posts/E9EevrzBcDMap6dbs/the-thingness-of-things",0,"",""],["Thoughts On Expanding the AI Safety Community: Benefits and Challenges of Outreach to Non-Technical Professionals","Yashvardhan Sharma","2023","blog","LessWrong","www.lesswrong.com/posts/4QgHqN2fHvqAwwSRg/thoughts-on-expanding-the-ai-safety-community-benefits-and",0,"","governance"],["Visualizing what ConvNets learn","Andrej Karpathy","2023","report","cs231n.github.io","cs231n.github.io/understanding-cnn/",0,"",""],["Would it be good or bad for the US military to get involved in AI risk?","Grant Demaree","2023","blog","LessWrong","www.lesswrong.com/posts/dTWevKRiMM4ptcjjg/would-it-be-good-or-bad-for-the-us-military-to-get-involved",0,"","governance robustness"],["'simulator' framing and confusions about LLMs","Beth Barnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dYnHLWMXCYdm9xu5j/simulator-framing-and-confusions-about-llms",0,"",""],["200 COP in MI: Interpreting Algorithmic Problems","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ejtFsvyhRkMofKAFy/200-cop-in-mi-interpreting-algorithmic-problems",0,"","interpretability mechanistic-interpretability"],["Are Mixture-of-Experts Transformers More Interpretable Than Dense Transformers?","simeon_c","2022","blog","LessWrong","www.lesswrong.com/posts/umkzHoeH2SJ2EcH6K/are-mixture-of-experts-transformers-more-interpretable-than",0,"","interpretability"],["end of 2022: my life so far","Tamsin Leake","2022","blog","carado.moe","carado.moe/my-life-so-far.html",0,"",""],["Racing through a minefield: the AI deployment problem","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XRphCh6NbfQiDF3Nt/racing-through-a-minefield-the-ai-deployment-problem",0,"","evals governance"],["Self-Limiting AI in AI Alignment","The_Lord's_Servant_280","2022","blog","EA Forum","forum.effectivealtruism.org/posts/B6FXBZBsBB2mmyp3z/self-limiting-ai-in-ai-alignment",0,"",""],["Should AI systems have to identify themselves?","Darren McKee","2022","blog","LessWrong","www.lesswrong.com/posts/SBPrRQYHyKFthZdRH/should-ai-systems-have-to-identify-themselves",0,"","governance"],["Beyond Rewards and Values: A Non-dualistic Approach to Universal Intelligence","Akira Pyinya","2022","blog","LessWrong","www.lesswrong.com/posts/NKbF8RvNiQyfWoz8e/beyond-rewards-and-values-a-non-dualistic-approach-to",0,"","theory"],["But is it really in Rome? An investigation of the ROME model editing technique","jacquesthibs","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QL7J9wmS6W2fWpofd/but-is-it-really-in-rome-an-investigation-of-the-rome-model",0,"","interpretability"],["Future Matters #6: FTX collapse, value lock-in, and counterarguments to AI x-risk","Pablo and matthew.vandermerwe","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tGpwWsP5iBfZFigeZ/future-matters-6-ftx-collapse-value-lock-in-and",0,"",""],["Models Don't \"Get Reward\"","Sam Ringer","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TWorNr22hhYegE4RT/models-don-t-get-reward",0,"",""],["My thoughts on OpenAI's alignment plan","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/FBG7AghvvP7fPYzkx/my-thoughts-on-openai-s-alignment-plan-1",0,"","automated-alignment-research governance"],["200 COP in MI: Looking for Circuits in the Wild","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XNjRwEX9kxbpzWFWd/200-cop-in-mi-looking-for-circuits-in-the-wild",0,"","interpretability mechanistic-interpretability"],["CFP for Rebellion and Disobedience in AI workshop","Ram Rachum","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GyZR24j8buR7D2n8D/cfp-for-rebellion-and-disobedience-in-ai-workshop",0,"",""],["Internal Interfaces Are a High-Priority Interpretability Target","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nwLQt4e7bstCyPEXs/internal-interfaces-are-a-high-priority-interpretability",0,"","interpretability"],["The commercial incentive to intentionally train AI to deceive us","Derek M. Jones","2022","blog","LessWrong","www.lesswrong.com/posts/gbaat54g4h6AA9pof/the-commercial-incentive-to-intentionally-train-ai-to",0,"","alignment-faking deception"],["200 Concrete Open Problems in Mechanistic Interpretability: Introduction","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LbrPTJ4fmABEdEnLf/200-concrete-open-problems-in-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability"],["200 COP in MI: The Case for Analysing Toy Language Models","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GWCgZrzWCZCuzGktv/200-cop-in-mi-the-case-for-analysing-toy-language-models",0,"","interpretability"],["Book recommendations for the history of ML?","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/SR3tkgMAswNc6jXvL/book-recommendations-for-the-history-of-ml",0,"",""],["Getting up to Speed on the Speed Prior in 2022","robertzk","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bC5xd7wQCnTDw7Kyx/getting-up-to-speed-on-the-speed-prior-in-2022",0,"","alignment-faking deception"],["In Defense of Wrapper-Minds","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wdC8fH8kHffYn3kNa/in-defense-of-wrapper-minds",0,"",""],["making decisions as our approximately simulated selves","Tamsin Leake","2022","blog","carado.moe","carado.moe/approximate-decisions.html",0,"",""],["Reflections on my 5-month AI alignment upskilling grant","Jay Bailey","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DnMg5q4Wyuuf99kkX/reflections-on-my-5-month-ai-alignment-upskilling-grant",0,"",""],["What AI Safety Materials Do ML Researchers Find Compelling?","Vael Gates and Collin","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gpk8dARHBi7Mkmzt9/what-ai-safety-materials-do-ml-researchers-find-compelling",0,"",""],["Can we efficiently distinguish different mechanisms?","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JLyWP2Y9LAruR2gi9/can-we-efficiently-distinguish-different-mechanisms",0,"","interpretability eliciting-latent-knowledge"],["How to Catch a ChatGPT Cheat: 7 Practical Tips","Marshall","2022","blog","EA Forum","forum.effectivealtruism.org/posts/brFTTy47YdGxCDzqp/how-to-catch-a-chatgpt-cheat-7-practical-tips",0,"",""],["I have thousands of copies of HPMOR in Russian. How to use them with the most impact?","Samin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AmA9gQMhqAQW8bC4W/i-have-thousands-of-copies-of-hpmor-in-russian-how-to-use",0,"",""],["Institutions Cannot Restrain Dark-Triad AI Exploitation","Remmelt and flandry19","2022","blog","LessWrong","www.lesswrong.com/posts/wGstGErtRegAzBjz9/institutions-cannot-restrain-dark-triad-ai-exploitation",0,"","governance"],["My Reservations about Discovering Latent Knowledge (Burns, Ye, et al)","Robert_AIZI","2022","blog","LessWrong","www.lesswrong.com/posts/C53REQuzSk3TTfqgT/my-reservations-about-discovering-latent-knowledge-burns-ye",0,"","eliciting-latent-knowledge"],["Reflections on my 5-month alignment upskilling grant","Jay Bailey","2022","blog","LessWrong","www.lesswrong.com/posts/wnF9iydYiBMRs2jPg/reflections-on-my-5-month-alignment-upskilling-grant",0,"",""],["The AIA and its Brussels Effect","Kathryn O'Rourke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/n8r2GWz5gSHn9dnob/the-aia-and-its-brussels-effect",0,"","evals governance policy"],["Why The Focus on Expected Utility Maximisers?","DragonGod","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XYDsYSbBjqgPAgcoQ/why-the-focus-on-expected-utility-maximisers",0,"",""],["Air-gapping evaluation and support","Ryan Kidd","2022","blog","LessWrong","www.lesswrong.com/posts/ehLR9HeXB5TMp9Y4v/air-gapping-evaluation-and-support",0,"","evals"],["An overview of some promising work by junior alignment researchers","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/jcFSEbXEfKgMwETqw/an-overview-of-some-promising-work-by-junior-alignment",0,"",""],["Analogies between Software Reverse Engineering and Mechanistic Interpretability","Neel Nanda and Itay Yona","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tsYcsZAkKsqLXC3Bu/analogies-between-software-reverse-engineering-and",0,"","interpretability mechanistic-interpretability"],["Announcing: The Independent AI Safety Registry","Shoshannah Tekofsky","2022","blog","LessWrong","www.lesswrong.com/posts/mXaugZyivQN3Eg8G3/announcing-the-independent-ai-safety-registry",0,"",""],["Avoiding perpetual risk from TAI","scasper","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FfTxEf3uFPsZf9EMP/avoiding-perpetual-risk-from-tai",0,"","governance"],["Coherent extrapolated dreaming","Alex Flint","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vdXNPzuh3fwgykvKY/coherent-extrapolated-dreaming",0,"",""],["How long till Brussels?: A light investigation into the Brussels Gap","Yadav","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QjeruoGQmYZh2ZsCt/how-long-till-brussels-a-light-investigation-into-the",0,"","governance policy"],["Mlyyrczo","lsusr","2022","blog","LessWrong","www.lesswrong.com/posts/GqyQSwYrryc4e2hgf/mlyyrczo",0,"",""],["Slightly against aligning with neo-luddites","Matthew_Barnett","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3yojNGhTXAydhfkNg/slightly-against-aligning-with-neo-luddites",0,"","evals governance policy"],["[Hebbian Natural Abstractions] Mathematical Foundations","Samuel Nellessen and Jan","2022","blog","LessWrong","www.lesswrong.com/posts/EASv46FpehppAFHSm/hebbian-natural-abstractions-mathematical-foundations",0,"",""],["Accurate Models of AI Risk Are Hyperexistential Exfohazards","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/xAzKefLsYdFa4SErg/accurate-models-of-ai-risk-are-hyperexistential-exfohazards",0,"","governance"],["Concrete Steps to Get Started in Transformer Mechanistic Interpretability","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9ezkEb9oGvEi6WoB3/concrete-steps-to-get-started-in-transformer-mechanistic",0,"","interpretability mechanistic-interpretability"],["I've updated towards AI boxing being surprisingly easy","Noosphere89","2022","blog","LessWrong","www.lesswrong.com/posts/osmwiGkCGxqPfLf4A/i-ve-updated-towards-ai-boxing-being-surprisingly-easy",0,"",""],["Oracle AGI - How can it escape, other than security issues? (Steganography?)","RationalSieve","2022","blog","LessWrong","www.lesswrong.com/posts/kG2qQGnpvecGg4dyn/oracle-agi-how-can-it-escape-other-than-security-issues",0,"",""],["Take 14: Corrigibility isn't that great.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/MWBFnH225LgRxfHgd/take-14-corrigibility-isn-t-that-great",0,"",""],["Is Eric Schmidt funding AI capabilities research by the US government?","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/aupKXpPGnFmbfE2xC/is-eric-schmidt-funding-ai-capabilities-research-by-the-us",0,"","governance policy"],["List #2: Why coordinating to align as humans to not develop AGI is a lot easier than, well... coordinating as humans with AGI coordinating to be aligned with humans","Remmelt","2022","blog","LessWrong","www.lesswrong.com/posts/SKyzGTbEEuNpoSFKH/list-2-why-coordinating-to-align-as-humans-to-not-develop",0,"","governance"],["Löb's Lemma: an easier approach to Löb's Theorem","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AkGvmJ6WE5sXjuwnC/loeb-s-lemma-an-easier-approach-to-loeb-s-theorem",0,"",""],["Practical AI risk I: Watching large compute","Gustavo Ramires","2022","blog","LessWrong","www.lesswrong.com/posts/zdKrgxwhE5pTiDpDm/practical-ai-risk-i-watching-large-compute",0,"","governance"],["Three reasons to cooperate","paulfchristiano","2022","blog","LessWrong","www.lesswrong.com/posts/mm8sFBpPH3Bb2NhGg/three-reasons-to-cooperate",0,"","theory"],["Katja Grace: Let's think about slowing down AI","peterhartree","2022","blog","EA Forum","forum.effectivealtruism.org/posts/sFemFbiFTntgtQDbD/katja-grace-let-s-think-about-slowing-down-ai",0,"","governance"],["Why is \"Argument Mapping\" Not More Common in EA/Rationality (And What Objections Should I Address in a Post on the Topic?)","Harrison Durland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HmYfoKW6FuyFHmwcJ/why-is-argument-mapping-not-more-common-in-ea-rationality",0,"",""],["Article Review: Discovering Latent Knowledge (Burns, Ye, et al)","Robert_AIZI","2022","blog","LessWrong","www.lesswrong.com/posts/WtKGLJQfjCWTm7tFK/article-review-discovering-latent-knowledge-burns-ye-et-al",0,"","eliciting-latent-knowledge"],["being only polynomial capabilities away from alignment: what a great problem to have that would be!","Tamsin Leake","2022","blog","carado.moe","carado.moe/capabilities-away-great-problem.html",0,"",""],["December 2022 updates and fundraising","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/december-2022-updates-and-fundraising/",0,"",""],["Let’s think about slowing down AI","KatjaGrace","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/uFNgRumrDTpBfQGrs/let-s-think-about-slowing-down-ai",0,"","governance"],["Let’s think about slowing down AI","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/lets-think-about-slowing-down-ai/",0,"",""],["one-shot AI, delegating embedded agency and decision theory, and one-shot QACI","Tamsin Leake","2022","blog","carado.moe","carado.moe/delegated-embedded-agency-decision-theory.html",0,"","theory"],["Racing through a minefield: the AI deployment problem","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/racing-through-a-minefield-the-ai-deployment-problem/",0,"",""],["Response to Holden’s alignment plan","Alex Flint","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CBvebC9FgSMtsD5T9/response-to-holden-s-alignment-plan",0,"",""],["Some Notes on the mathematics of Toy Autoencoding Problems","Spencer Becker-Kahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dodEwdiphp5TJZjJj/some-notes-on-the-mathematics-of-toy-autoencoding-problems",0,"","interpretability"],["Take 13: RLHF bad, conditioning good.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AXpXG9oTiucidnqPK/take-13-rlhf-bad-conditioning-good",0,"","rlhf"],["[DISC] Are Values Robust?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/YoFLKyTJ7o4ApcKXR/disc-are-values-robust",0,"",""],["A Comprehensive Mechanistic Interpretability Explainer & Glossary","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vnocLyeWXcAxtdDnP/a-comprehensive-mechanistic-interpretability-explainer-and",0,"","interpretability mechanistic-interpretability"],["Applications Open: GovAI Summer Fellowship 2023","GovAI","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pDjtcoawgvpDpoyrL/applications-open-govai-summer-fellowship-2023",0,"","governance compute-governance"],["Background for \"Understanding the diffusion of large language models\"","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/oB3MnFQa8LqcuEhjG/background-for-understanding-the-diffusion-of-large-language",0,"","governance forecasting"],["CIRL Corrigibility is Fragile","rachelAF and AdamGleave","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PGK3AJtNG4rPHuZxy/cirl-corrigibility-is-fragile",0,"",""],["Conclusion and Bibliography for \"Understanding the diffusion of large language models\"","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pZPDQmyEoaqBB8szD/conclusion-and-bibliography-for-understanding-the-diffusion",0,"","governance forecasting"],["Decisions: Ontologically Shifting to Determinism","Chris_Leong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dMvtDTdQTRTjxuvfE/decisions-ontologically-shifting-to-determinism",0,"","theory"],["Drivers of large language model diffusion: incremental research, publicity, and cascades","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eHKLgsXfMvSyAWb7E/drivers-of-large-language-model-diffusion-incremental",0,"","governance forecasting"],["GPT-3-like models are now much easier to access and deploy than to develop","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/foptmf8C25TzJuit6/gpt-3-like-models-are-now-much-easier-to-access-and-deploy",0,"","governance forecasting"],["Implications of large language model diffusion for AI governance","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/suBJdDkEu9EaSmTxJ/implications-of-large-language-model-diffusion-for-ai",0,"","governance compute-governance forecasting"],["New AI risk intro from Vox [link post]","JakubK","2022","blog","LessWrong","www.lesswrong.com/posts/sfhNLrCdDwsc6zkQa/new-ai-risk-intro-from-vox-link-post",0,"",""],["Publication decisions for large language models, and their impacts","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KkbEfpNkjNepQrj8g/publication-decisions-for-large-language-models-and-their",0,"","governance forecasting"],["Questions for further investigation of AI diffusion","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/4PAi6nNRfQwwhdtBW/questions-for-further-investigation-of-ai-diffusion",0,"","governance compute-governance forecasting"],["The replication and emulation of GPT-3","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FANYsqzPM9Yht3KM2/the-replication-and-emulation-of-gpt-3",0,"","governance forecasting"],["the scarcity of moral patient involvement","Tamsin Leake","2022","blog","carado.moe","carado.moe/scarce-moral-patient-involvement.html",0,"",""],["Understanding the diffusion of large language models: summary","Ben Cottier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nc3JFZbqnzWWAPkmz/understanding-the-diffusion-of-large-language-models-summary-1",0,"","governance compute-governance forecasting"],["An Open Agency Architecture for Safe Transformative AI","davidad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pKSmEkSQJsCSTK6nH/an-open-agency-architecture-for-safe-transformative-ai",0,"",""],["Discovering Language Model Behaviors with Model-Written Evaluations","evhub and Ethan Perez","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yRAo2KEGWenKYZG9K/discovering-language-model-behaviors-with-model-written",0,"","evals"],["High-level hopes for AI alignment","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/rJRw78oihoT5paFGd/high-level-hopes-for-ai-alignment",0,"","interpretability evals"],["Note on algorithms with multiple trained components","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hRuE66SXobGhrbwxR/note-on-algorithms-with-multiple-trained-components",0,"","reward-hacking"],["Posit: Most AI safety people should work on alignment/safety challenges for AI tools that already have users (Stable Diffusion, GPT)","nonzerosum","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MfPWk4ToW3p6utWpc/posit-most-ai-safety-people-should-work-on-alignment-safety",0,"",""],["Take 12: RLHF's use is evidence that orgs will jam RL at real-world problems.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QQMzxSJDgWkAhupi5/take-12-rlhf-s-use-is-evidence-that-orgs-will-jam-rl-at-real",0,"","rlhf"],["The \"Minimal Latents\" Approach to Natural Abstractions","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/N2JcFZ3LCCsnK2Fep/the-minimal-latents-approach-to-natural-abstractions",0,"",""],["AGI Timelines in Governance: Different Strategies for Different Timeframes","simeon_c and AmberDawn","2022","blog","LessWrong","www.lesswrong.com/posts/6CjnFcsRHJesR9MEA/agi-timelines-in-governance-different-strategies-for",0,"","governance forecasting"],["Conditions for Superrationality-motivated Cooperation in a one-shot Prisoner's Dilemma","Jim Buhler","2022","blog","LessWrong","www.lesswrong.com/posts/HLXiJgqxuMpwamdar/conditions-for-superrationality-motivated-cooperation-in-a",0,"","theory"],["Discovering Language Model Behaviors with Model-Written Evaluations","Ethan Perez and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2212.09251",0,"","rlhf evals sycophancy power-seeking robustness"],["Event [Berkeley]: Alignment Collaborator Speed-Meeting","AlexMennen and Carson Jones","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hQZcoSBeAHLSzxhYi/event-berkeley-alignment-collaborator-speed-meeting",0,"",""],["our deepest wishes","Tamsin Leake","2022","blog","carado.moe","carado.moe/our-deepest-wishes.html",0,"",""],["Results from a survey on tool use and workflows in alignment research","jacquesthibs and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/a2io2mcxTWS4mxodF/results-from-a-survey-on-tool-use-and-workflows-in-alignment",0,"","tool-use automated-alignment-research"],["Shard Theory in Nine Theses: a Distillation and Critical Appraisal","LawrenceC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8ccTZ9ZxpJrvnxt4F/shard-theory-in-nine-theses-a-distillation-and-critical",0,"",""],["The ‘Old AI’: Lessons for AI governance from early electricity regulation","Sam Clarke and Di Cooke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/k73qrirnxcKtKZ4ng/the-old-ai-lessons-for-ai-governance-from-early-electricity-1",0,"","evals governance policy"],["Towards Hodge-podge Alignment","Cleo Nardo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YnGRBADQwpYRbuCbz/towards-hodge-podge-alignment-1",0,"",""],["Why I think that teaching philosophy is high impact","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9cCyPE2EDpjpJvqnF/why-i-think-that-teaching-philosophy-is-high-impact",0,"",""],["Why I think that teaching philosophy is high impact","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/FnLt23WFhkSPT9Dgc/why-i-think-that-teaching-philosophy-is-high-impact",0,"",""],["Will research in AI risk jinx it? Consequences of training AI on AI risk arguments","Yann Dubois","2022","blog","LessWrong","www.lesswrong.com/posts/evtJJeghGM5aAM5W7/will-research-in-ai-risk-jinx-it-consequences-of-training-ai",0,"","rlhf"],["Take 11: \"Aligning language models\" should be weirder.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tYbusKv4Yci3GaBiM/take-11-aligning-language-models-should-be-weirder",0,"",""],["Looking for an alignment tutor","JanBrauner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3us74zNBGgFJeAXTo/looking-for-an-alignment-tutor",0,"",""],["Positive values seem more robust and lasting than prohibitions","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cHnQ4bBFr3cX6rBxh/positive-values-seem-more-robust-and-lasting-than",0,"",""],["There have been 3 planes (billionaire donors) and 2 have crashed","trevor1","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XSs6HqFvTAHR3bHLg/there-have-been-3-planes-billionaire-donors-and-2-have",0,"",""],["There have been 3 planes (billionaire donors) and 2 have crashed","trevor","2022","blog","LessWrong","www.lesswrong.com/posts/tMvw3HiYB6oKbPX8m/there-have-been-3-planes-billionaire-donors-and-2-have",0,"",""],["What we owe the microbiome","TeddyW","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DdzSEFBEb6rtfChpN/what-we-owe-the-microbiome",0,"",""],["AI overhangs depend on whether algorithms, compute and data are substitutes or complements","NathanBarnard","2022","blog","LessWrong","www.lesswrong.com/posts/X3z3rtzGG6F4ZWADQ/ai-overhangs-depend-on-whether-algorithms-compute-and-data",0,"","forecasting"],["Can we efficiently explain model behaviors?","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dQvxMZkfgqGitWdkb/can-we-efficiently-explain-model-behaviors",0,"","interpretability eliciting-latent-knowledge"],["Concrete actionable policies relevant to AI safety (written 2019)","weeatquince","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xGTcoL4rJsxGuDLFy/concrete-actionable-policies-relevant-to-ai-safety-written",0,"","governance policy"],["How important are accurate AI timelines for the optimal spending schedule on AI risk interventions?","Tristan Cook","2022","blog","EA Forum","forum.effectivealtruism.org/posts/boxF7ZL5zLieFLCtv/how-important-are-accurate-ai-timelines-for-the-optimal",0,"","forecasting"],["How would you estimate the value of delaying AGI by 1 day, in marginal donations to GiveWell?","AnonymousAccount","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KDtg6dzjcETnJGaQr/how-would-you-estimate-the-value-of-delaying-agi-by-1-day-in",0,"","policy forecasting"],["Paper: Constitutional AI: Harmlessness from AI Feedback (Anthropic)","LawrenceC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/aLhLGns2BSun3EzXB/paper-constitutional-ai-harmlessness-from-ai-feedback",0,"","rlhf constitutional-ai"],["Paper: Transformers learn in-context by gradient descent","LawrenceC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/firtXAWGdvzXYAh9B/paper-transformers-learn-in-context-by-gradient-descent",0,"","interpretability"],["Point-E: A system for generating 3D point clouds from complex prompts","OpenAI Research","2022","blog","openai.com","openai.com/research/point-e",0,"",""],["Proper scoring rules don’t guarantee predicting fixed points","Johannes Treutlein and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Aufg88v7mQ2RuEXkS/proper-scoring-rules-don-t-guarantee-predicting-fixed-points",0,"","theory"],["We should say more than “x-risk is high”","OllieBase","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cGM86RhxMdfDYbQnn/we-should-say-more-than-x-risk-is-high",0,"",""],["Who will be in charge once alignment is achieved?","trurl","2022","blog","EA Forum","forum.effectivealtruism.org/posts/H9uPyi6MGmzer5i9b/who-will-be-in-charge-once-alignment-is-achieved",0,"","policy"],["AI Neorealism: a threat model & success criterion for existential safety","davidad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Wi4RAJCbh3qD9fynj/ai-neorealism-a-threat-model-and-success-criterion-for",0,"",""],["AI Safety Movement Builders should help the community to optimise three factors: contributors, contributions and coordination","peterslattery","2022","blog","LessWrong","www.lesswrong.com/posts/Fz2Sdh24RjaaMkQRW/ai-safety-movement-builders-should-help-the-community-to",0,"",""],["High-level hopes for AI alignment","HoldenKarnofsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/7BWmLhFtqzqEPs8d5/high-level-hopes-for-ai-alignment",0,"",""],["High-level hopes for AI alignment","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/high-level-hopes-for-ai-alignment/",0,"",""],["How \"Discovering Latent Knowledge in Language Models Without Supervision\" Fits Into a Broader Alignment Scheme","Collin","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/L4anhrxjv8j2yRKKp/how-discovering-latent-knowledge-in-language-models-without",0,"","interpretability eliciting-latent-knowledge"],["how far are things that care?","Tamsin Leake","2022","blog","carado.moe","carado.moe/how-far-are-things-that-care.html",0,"",""],["How is ARC planning to use ELK?","jacquesthibs","2022","blog","LessWrong","www.lesswrong.com/posts/tz4ZGANPxADmmt8hS/how-is-arc-planning-to-use-elk",0,"","eliciting-latent-knowledge"],["Part 2: AI Safety Movement Builders should help the community to optimise three factors: contributors, contributions and coordination","PeterSlattery","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YMvSZi2EWxNHwFtbb/part-2-ai-safety-movement-builders-should-help-the-community",0,"",""],["The next decades might be wild","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qRtD4WqKRYEtT5pi3/the-next-decades-might-be-wild",0,"",""],["all claw, no world — and other thoughts on the universal distribution","Tamsin Leake","2022","blog","carado.moe","carado.moe/all-claw-no-world.html",0,"",""],["Discovering Latent Knowledge in Language Models Without Supervision","Xodarap","2022","blog","LessWrong","www.lesswrong.com/posts/kCEjcu53EEiqBH4gN/discovering-latent-knowledge-in-language-models-without",0,"","eliciting-latent-knowledge"],["EA's Achievements in 2022","ElliotJDavies","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pzCi5EuiherL2ccYc/ea-s-achievements-in-2022",0,"","policy"],["Extracting and Evaluating Causal Direction in LLMs' Activations","Fabien Roger and simeon_c","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mtZEjDortt8E8Apjs/extracting-and-evaluating-causal-direction-in-llms",0,"","interpretability evals"],["Is the AI timeline too short to have children?","Yoreth","2022","blog","LessWrong","www.lesswrong.com/posts/wx793JXieh97AXtg6/is-the-ai-timeline-too-short-to-have-children",0,"","forecasting"],["My AGI safety research—2022 review, ’23 plans","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qusBXzCpxijTudvBB/my-agi-safety-research-2022-review-23-plans",0,"",""],["Predicting GPU performance","Marius Hobbhahn and Tamay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vnvGhfikBbrjZHMuD/predicting-gpu-performance",0,"",""],["Seeking participants for study of AI safety researchers","Gardner","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9AqZL4FnhP7wfgjoM/seeking-participants-for-study-of-ai-safety-researchers",0,"",""],["Trying to disambiguate different questions about whether RLHF is “good”","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NG6FrXgmqPd5Wn3mh/trying-to-disambiguate-different-questions-about-whether",0,"","rlhf"],["«Boundaries», Part 3b: Alignment problems in terms of boundaries","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SajYfrsoTHxiXPNtf/boundaries-part-3b-alignment-problems-in-terms-of-boundaries",0,"",""],["[Interim research report] Taking features out of superposition with sparse autoencoders","Lee Sharkey and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/z6QQJbtpkEAX3Aojj/interim-research-report-taking-features-out-of-superposition",0,"","interpretability mechanistic-interpretability"],["AI alignment is distinct from its near-term applications","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hw26MrLuhGWH7kBLm/ai-alignment-is-distinct-from-its-near-term-applications",0,"",""],["Alignment with argument-networks and assessment-predictions","Tor Økland Barstad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gvm8WvLmazj2NdTPk/alignment-with-argument-networks-and-assessment-predictions",0,"","automated-alignment-research"],["An exploration of GPT-2's embedding weights","Adam Scherlis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BMghmAxYxeSdAteDc/an-exploration-of-gpt-2-s-embedding-weights",0,"","interpretability"],["Applications open for AGI Safety Fundamentals: Alignment Course","Richard_Ngo and Jamie Bernardi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mXKjuquNC8ivpiKWz/applications-open-for-agi-safety-fundamentals-alignment-1",0,"",""],["Applications open for AGI Safety Fundamentals: Alignment Course","Jamie Bernardi and richard_ngo","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HBgAruFrZhFKBFfDa/applications-open-for-agi-safety-fundamentals-alignment",0,"",""],["Are lawsuits against AGI companies extending AGI timelines?","SlowingAGI","2022","blog","LessWrong","www.lesswrong.com/posts/6omuuguhMLxFC3Sah/are-lawsuits-against-agi-companies-extending-agi-timelines",0,"","forecasting"],["Best introductory overviews of AGI safety?","JakubK","2022","blog","LessWrong","www.lesswrong.com/posts/T98kdFL5bxBWSiE3N/best-introductory-overviews-of-agi-safety",0,"",""],["Existential AI Safety is NOT separate from near-term applications","scasper","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yKzyCw5EjabyZRkbJ/existential-ai-safety-is-not-separate-from-near-term",0,"","governance"],["Limits of Superintelligence","Aleksei Petrenko","2022","blog","LessWrong","www.lesswrong.com/posts/Ck5ywHRHAjMmSoomy/limits-of-superintelligence",0,"",""],["Take 10: Fine-tuning with RLHF is aesthetically unsatisfying.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QujNmRy3uFyrkfqb7/take-10-fine-tuning-with-rlhf-is-aesthetically-unsatisfying",0,"","rlhf"],["12 career-related questions that may (or may not) be helpful for people interested in alignment research","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/Q37Ay82dfb3wnKjTr/12-career-related-questions-that-may-or-may-not-be-helpful",0,"",""],["Concept extrapolation for hypothesis generation","Stuart_Armstrong and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jjiDARiv4hybZXeXL/concept-extrapolation-for-hypothesis-generation",0,"",""],["Join the AI Testing Hackathon this Friday","Esben Kran and Apart Research","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JE3ZjEoWot6yQFSJj/join-the-ai-testing-hackathon-this-friday",0,"",""],["Side-channels: input versus output","davidad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bqRD6MS3yCdfM9wRe/side-channels-input-versus-output",0,"",""],["Take 9: No, RLHF/IDA/debate doesn't solve outer alignment.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6YNZt5xbBT5dJXknC/take-9-no-rlhf-ida-debate-doesn-t-solve-outer-alignment",0,"","rlhf"],["a rough sketch of formal aligned AI using QACI","Tamsin Leake","2022","blog","carado.moe","carado.moe/rough-sketch-formal-aligned-ai.html",0,"",""],["AI Safety Seems Hard to Measure","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NbiHKTN5QhFFfjjm5/ai-safety-seems-hard-to-measure",0,"","evals"],["An appraisal of the Future of Life Institute AI existential risk program","PabloAMC","2022","blog","EA Forum","forum.effectivealtruism.org/posts/44XPFrHiFwFBM2jfL/an-appraisal-of-the-future-of-life-institute-ai-existential",0,"",""],["Benchmarks for Comparing Human and AI Intelligence","ViktorThink","2022","blog","LessWrong","www.lesswrong.com/posts/cDFj427x9LzgsMKv4/benchmarks-for-comparing-human-and-ai-intelligence",0,"","benchmarks forecasting"],["Finite Factored Sets in Pictures","Magdalena Wache","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PfcQguFpT8CDHcozj/finite-factored-sets-in-pictures-6",0,"",""],["Please provide feedback on AI-safety grant proposal, thanks!","Alex Long","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MEEXNgCDTKccmWpmY/please-provide-feedback-on-ai-safety-grant-proposal-thanks",0,"",""],["Reflections on the PIBBSS Fellowship 2022","Nora_Ammann and particlemania","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gbeyjALdjdoCGayc6/reflections-on-the-pibbss-fellowship-2022",0,"",""],["Reflections on the PIBBSS Fellowship 2022","nora and particlemania","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zvALRCKshYGYetsbC/reflections-on-the-pibbss-fellowship-2022",0,"",""],["Reframing inner alignment","davidad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xMQ7vwFACQX3gZouv/reframing-inner-alignment",0,"",""],["[ASoT] Natural abstractions and AlphaZero","Ulisse Mini","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/obht9QqMDMNLwhPQS/asot-natural-abstractions-and-alphazero",0,"","interpretability"],["ChatGPT can write code! ?","Miguel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DBaLPBcWyQtY34Kt9/chatgpt-can-write-code",0,"",""],["Cooperation, Avoidance, and Indifference: Alternate Futures for Misaligned AGI","Kiel Brennan-Marquez","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3K2fKB8azNoEiEL9t/cooperation-avoidance-and-indifference-alternate-futures-for",0,"",""],["How promising are legal avenues to restrict AI training data?","thehalliard","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nCbAHnpi4LGFR32yq/how-promising-are-legal-avenues-to-restrict-ai-training-data",0,"","governance training-data"],["My thoughts on OpenAI's Alignment plan","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3oNZA9wTrFJRH6Sau/my-thoughts-on-openai-s-alignment-plan",0,"",""],["Announcing BlueDot Impact","Dewi and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3EWpLid8tkyYJakfm/announcing-bluedot-impact",0,"",""],["Fear mitigated the nuclear threat, can it do the same to AGI risks?","Igor Ivanov","2022","blog","LessWrong","www.lesswrong.com/posts/2yjoEKE9ryuCitBRs/fear-mitigated-the-nuclear-threat-can-it-do-the-same-to-agi",0,"",""],["ML Safety at NeurIPS & Paradigmatic AI Safety? MLAISU W49","Esben Kran and Steinthal","2022","blog","LessWrong","www.lesswrong.com/posts/3cgevkQRAjSPdynJw/ml-safety-at-neurips-and-paradigmatic-ai-safety-mlaisu-w49",0,"",""],["Prosaic misalignment from the Solomonoff Predictor","Cleo Nardo","2022","blog","LessWrong","www.lesswrong.com/posts/GfFvsPaSFG7wqY4sk/prosaic-misalignment-from-the-solomonoff-predictor",0,"","agents"],["Take 8: Queer the inner/outer alignment dichotomy.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dyg4KSyMJC8cNDMK6/take-8-queer-the-inner-outer-alignment-dichotomy",0,"",""],["Working towards AI alignment is better","Johannes C. Mayer","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pnqkGiGcshtgF2fnQ/working-towards-ai-alignment-is-better",0,"",""],["You can still fetch the coffee today if you're dead tomorrow","davidad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dzDKDRJPQ3kGqfER9/you-can-still-fetch-the-coffee-today-if-you-re-dead-tomorrow",0,"","instrumental-convergence"],["AI Safety Seems Hard to Measure","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/ai-safety-seems-hard-to-measure/",0,"",""],["AI Safety Seems Hard to Measure","HoldenKarnofsky","2022","blog","LessWrong","www.lesswrong.com/posts/7gkXuHEm6CqEGT2mg/ai-safety-seems-hard-to-measure",0,"",""],["I Believe we are in a Hardware Overhang","nem","2022","blog","LessWrong","www.lesswrong.com/posts/JfkLHWJsFtk9LHhgR/i-believe-we-are-in-a-hardware-overhang",0,"","forecasting"],["If Wentworth is right about natural abstractions, it would be bad for alignment","Wuschel Schulz","2022","blog","LessWrong","www.lesswrong.com/posts/nJHXQWCSByS4SxfQz/if-wentworth-is-right-about-natural-abstractions-it-would-be",0,"",""],["Main paths to impact in EU AI Policy","JOMG_Monnet","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wPHpdwfu3toRDf6hM/main-paths-to-impact-in-eu-ai-policy",0,"","evals governance policy"],["Notes on OpenAI’s alignment plan","Alex Flint","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FTk7ufqK2D4dkdBDr/notes-on-openai-s-alignment-plan",0,"","scalable-oversight"],["Riffing on the agent type","Quinn","2022","blog","LessWrong","www.lesswrong.com/posts/jEXdGBpD723DhizAZ/riffing-on-the-agent-type",0,"","agents theory"],["Take 7: You should talk about \"the human's utility function\" less.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hrYvdrqMyCnw3pBkd/take-7-you-should-talk-about-the-human-s-utility-function",0,"",""],["Why I'm Sceptical of Foom","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/jdLmC46ZuXS54LKzL/why-i-m-sceptical-of-foom",0,"","forecasting"],["Discovering Latent Knowledge in Language Models Without Supervision","Collin Burns","2022","paper","arXiv preprint","arxiv.org/abs/2212.03827",0,"","evals"],["Promoting compassionate longtermism","jonleighton","2022","blog","EA Forum","forum.effectivealtruism.org/posts/F2YfRtMvHfRJibwkj/promoting-compassionate-longtermism",0,"","governance policy"],["Simple Way to Prevent Power-Seeking AI","research_prime_space","2022","blog","LessWrong","www.lesswrong.com/posts/iozsJQ7fEdTCRxtJc/simple-way-to-prevent-power-seeking-ai",0,"","power-seeking"],["Something to make myself fascinated with computing science and AI.","Eduardo","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XcsX8GEkszEhEumMo/something-to-make-myself-fascinated-with-computing-science",0,"",""],["Take 6: CAIS is actually Orwellian.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CNrz9uy5Y4ELypzca/take-6-cais-is-actually-orwellian",0,"",""],["Thoughts on AGI organizations and capabilities work","Rob Bensinger and So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tuwwLQT4wqk25ndxk/thoughts-on-agi-organizations-and-capabilities-work",0,"",""],["Thoughts on AGI organizations and capabilities work","RobBensinger and So8res","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JFyzCv5YynN665nH8/thoughts-on-agi-organizations-and-capabilities-work",0,"",""],["AI for the board game Diplomacy","Yoram Bachrach and János Kramár","2022","blog","deepmind.com","www.deepmind.com/blog/ai-for-the-board-game-diplomacy",0,"",""],["AI Safety in a Vulnerable World: Requesting Feedback on Preliminary Thoughts","Jordan Arel","2022","blog","LessWrong","www.lesswrong.com/posts/iTmu5nrrtqHGe9iCr/ai-safety-in-a-vulnerable-world-requesting-feedback-on",0,"",""],["In defense of probably wrong mechanistic models","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/q5Gox77ReFAy5i2YQ/in-defense-of-probably-wrong-mechanistic-models",0,"",""],["Mesa-Optimizers via Grokking","orthonormal","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fytgZ26AgxmrAdyB4/mesa-optimizers-via-grokking",0,"",""],["Take 5: Another problem for natural abstractions is laziness.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/uR2uWMD9JGnRnYSeM/take-5-another-problem-for-natural-abstractions-is-laziness",0,"",""],["Using GPT-Eliezer against ChatGPT Jailbreaking","Stuart_Armstrong and rgorman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pNcFYZnPdXyL2RfgA/using-gpt-eliezer-against-chatgpt-jailbreaking",0,"","jailbreaks"],["Verification Is Not Easier Than Generation In General","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2PDC69DDJuAx6GANa/verification-is-not-easier-than-generation-in-general",0,"",""],["[Link] Why I’m optimistic about OpenAI’s alignment approach","janleike","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FAJWEfXxws8pMp8Hk/link-why-i-m-optimistic-about-openai-s-alignment-approach",0,"","rlhf automated-alignment-research"],["A Tentative Timeline of The Near Future (2022-2025) for Self-Accountability","Yitz","2022","blog","LessWrong","www.lesswrong.com/posts/oqJpxY2yZXg52QomP/a-tentative-timeline-of-the-near-future-2022-2025-for-self",0,"","forecasting"],["AI Safety Pitches post ChatGPT","ojorgensen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KegkXNBXoD7WKJtDk/ai-safety-pitches-post-chatgpt",0,"",""],["Aligned Behavior is not Evidence of Alignment Past a Certain Level of Intelligence","Ronny Fernandez","2022","blog","LessWrong","www.lesswrong.com/posts/XFt9ipeezjEqJCuY4/aligned-behavior-is-not-evidence-of-alignment-past-a-certain",0,"",""],["Analysis of AI Safety surveys for field-building insights","Ash Jafari","2022","blog","LessWrong","www.lesswrong.com/posts/4TCdZN2aj8rnuEkbH/analysis-of-ai-safety-surveys-for-field-building-insights",0,"",""],["ChatGPT on Spielberg’s A.I. and AI Alignment","Bill Benzon","2022","blog","LessWrong","www.lesswrong.com/posts/AbkzoSpad4XmHrh2Q/chatgpt-on-spielberg-s-a-i-and-ai-alignment",0,"",""],["Foresight for AGI Safety Strategy: Mitigating Risks and Identifying Golden Opportunities","jacquesthibs","2022","blog","LessWrong","www.lesswrong.com/posts/GbXAeq6smRzmYRSQg/foresight-for-agi-safety-strategy-mitigating-risks-and",0,"","governance forecasting"],["Have your timelines changed as a result of ChatGPT?","Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cYRHuqumCigYPHG6d/have-your-timelines-changed-as-a-result-of-chatgpt",0,"","forecasting"],["Is the \"Valley of Confused Abstractions\" real?","jacquesthibs","2022","blog","LessWrong","www.lesswrong.com/posts/g7rLyjg67iopg9zLD/is-the-valley-of-confused-abstractions-real",0,"","interpretability"],["Probably good projects for the AI safety ecosystem","Ryan Kidd","2022","blog","LessWrong","www.lesswrong.com/posts/v5z6rDuFPKM5dLpz8/probably-good-projects-for-the-ai-safety-ecosystem",0,"","robustness"],["Share your requests for ChatGPT","Kate Tran","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dmzfYk5HpxRuoQJmt/share-your-requests-for-chatgpt",0,"",""],["Steering Behaviour: Testing for (Non-)Myopia in Language Models","Evan R. Murphy and Megan Kinniment","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BuRt2igbFx9KaB5QG/steering-behaviour-testing-for-non-myopia-in-language-models",0,"","rlhf alignment-faking deception"],["Take 4: One problem with natural abstractions is there's too many of them.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NK4XxyrjFWt83m3dx/take-4-one-problem-with-natural-abstractions-is-there-s-too",0,"",""],["Updating my AI timelines","Matthew Barnett","2022","blog","LessWrong","www.lesswrong.com/posts/sbb9bZgojmEa7Yjrc/updating-my-ai-timelines",0,"","forecasting"],["AGI as a Black Swan Event","Stephen McAleese","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YKY4KmKEurY8cwHTJ/agi-as-a-black-swan-event",0,"",""],["AI can exploit safety plans posted on the Internet","Peter S. Park","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dzS6MwDdYcFFgmBFj/ai-can-exploit-safety-plans-posted-on-the-internet",0,"",""],["Race to the Top: Benchmarks for AI Safety","isaduan","2022","blog","EA Forum","forum.effectivealtruism.org/posts/saEXX9Nucz8mh9XgB/race-to-the-top-benchmarks-for-ai-safety",0,"","benchmarks governance policy"],["Race to the Top: Benchmarks for AI Safety","Isabella Duan","2022","blog","LessWrong","www.lesswrong.com/posts/KQ6fGiPeMnzzC6p9q/race-to-the-top-benchmarks-for-ai-safety",0,"","benchmarks"],["Take 3: No indescribable heavenworlds.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xpGFA7bdoiNDTut8C/take-3-no-indescribable-heavenworlds",0,"",""],["Causal Scrubbing: a method for rigorously testing interpretability hypotheses [Redwood Research]","LawrenceC and 7 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JvZhhzycHu2Yd57RN/causal-scrubbing-a-method-for-rigorously-testing",0,"","interpretability robustness"],["Causal scrubbing: Appendix","LawrenceC and 7 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kcZZAsEjwrbczxN2i/causal-scrubbing-appendix",0,"","interpretability robustness"],["Causal scrubbing: results on a paren balance checker","LawrenceC and 8 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kjudfaQazMmC74SbF/causal-scrubbing-results-on-a-paren-balance-checker",0,"","interpretability robustness"],["Causal scrubbing: results on induction heads","LawrenceC and 8 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/j6s9H9SHrEhEfuJnq/causal-scrubbing-results-on-induction-heads",0,"","interpretability robustness"],["Logical induction for software engineers","Alex Flint","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jtMXj24Masrnq3SpS/logical-induction-for-software-engineers",0,"","theory"],["Take 2: Building tools to help build FAI is a legitimate strategy, but it's dual-use.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pxiaLFjyr4WPmFdcm/take-2-building-tools-to-help-build-fai-is-a-legitimate",0,"",""],["Will the first AGI agent have been designed as an agent (in addition to an AGI)?","nahoj","2022","blog","LessWrong","www.lesswrong.com/posts/RsLsBnr6qmfKYR7sL/will-the-first-agi-agent-have-been-designed-as-an-agent-in",0,"","agents forecasting"],["[ASoT] Finetuning, RL, and GPT's world prior","Jozdien","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rh477a7fmWmzQdLMj/asot-finetuning-rl-and-gpt-s-world-prior",0,"","rlhf"],["Announcing the Cambridge Boston Alignment Initiative [Hiring!]","kuhanj and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xQBcrPsH57MjCcgTb/announcing-the-cambridge-boston-alignment-initiative-hiring",0,"",""],["Apply for the ML Winter Camp in Cambridge, UK [2-10 Jan]","Nathan_Barnard and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/G3vzNHjrL8AQmBqFb/apply-for-the-ml-winter-camp-in-cambridge-uk-2-10-jan",0,"",""],["Deconfusing Direct vs Amortised Optimization","beren","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/S54HKhxQyttNLATKu/deconfusing-direct-vs-amortised-optimization",0,"",""],["Inner and outer alignment decompose one hard problem into two extremely hard problems","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gHefoxiznGfsbiAu9/inner-and-outer-alignment-decompose-one-hard-problem-into",0,"",""],["Jailbreaking ChatGPT on Release Day","Zvi","2022","blog","LessWrong","www.lesswrong.com/posts/RYcoJdvmoBbi5Nax7/jailbreaking-chatgpt-on-release-day",0,"","jailbreaks"],["Subsets and quotients in interpretability","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ryv3FviYuovtJbgQd/subsets-and-quotients-in-interpretability",0,"","interpretability"],["Takeoff speeds, the chimps analogy, and the Cultural Intelligence Hypothesis","NickGabs","2022","blog","LessWrong","www.lesswrong.com/posts/xedQBnBR4dRtBkWpZ/takeoff-speeds-the-chimps-analogy-and-the-cultural",0,"","forecasting"],["[LINK] - ChatGPT discussion","JanBrauner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vnfPeiY3bwhaEMoXR/link-chatgpt-discussion",0,"",""],["A challenge for AGI organizations, and a challenge for readers","Rob Bensinger and Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tD9zEiHfkvakpnNam/a-challenge-for-agi-organizations-and-a-challenge-for-1",0,"",""],["Concrete actions to improve AI governance: the behaviour science approach","AlexanderSaeri","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LD6wKNdPbxfdgYnao/concrete-actions-to-improve-ai-governance-the-behaviour",0,"","governance"],["Distillation of \"How Likely is Deceptive Alignment?\"","NickGabs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HexzSqmfx9APAdKnh/distillation-of-how-likely-is-deceptive-alignment",0,"","alignment-faking deception"],["Finding gliders in the game of life","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FwYMuD2sNcaEpE5on/finding-gliders-in-the-game-of-life",0,"","interpretability eliciting-latent-knowledge"],["Re-Examining LayerNorm","Eric Winsor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jfG6vdJZCwTQmG7kb/re-examining-layernorm",0,"","interpretability"],["Research request (alignment strategy): Deep dive on \"making AI solve alignment for us\"","JanBrauner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/amoS8fGYsRKo6Wsdd/research-request-alignment-strategy-deep-dive-on-making-ai",0,"","automated-alignment-research"],["Take 1: We're not going to reverse-engineer the AI.","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/d52aS7jNcmi6miGbw/take-1-we-re-not-going-to-reverse-engineer-the-ai",0,"",""],["The Plan - 2022 Update","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BzYmJYECAc3xyCTt6/the-plan-2022-update",0,"","interpretability"],["Theories of impact for Science of Deep Learning","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tKYGvA9dKHa3GWBBk/theories-of-impact-for-science-of-deep-learning",0,"","interpretability"],["AI takeover tabletop RPG: \"The Treacherous Turn\"","Daniel Kokotajlo","2022","blog","LessWrong","www.lesswrong.com/posts/b5EqwQZw7ww2K28Ki/ai-takeover-tabletop-rpg-the-treacherous-turn",0,"",""],["Biological Anchors external review by Jennifer Lin (linkpost)","peterhartree","2022","blog","EA Forum","forum.effectivealtruism.org/posts/bFDwxxfErRStMvuAQ/biological-anchors-external-review-by-jennifer-lin-linkpost",0,"","forecasting"],["Compute Accounting Principles Can Help Reduce AI Risks","Krystal Jackson and 3 others","2022","report","techpolicy.press","techpolicy.press/compute-accounting-principles-can-help-reduce-ai-risks/",0,"",""],["Multi-Component Learning and S-Curves","Adam Jermyn and Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RKDQCB6smLWgs2Mhr/multi-component-learning-and-s-curves",0,"","interpretability"],["\"The Physicists\": A play about extinction and the responsibility of scientists","Lara_TH","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TaJrx7XHMdK6kvQ9X/the-physicists-a-play-about-extinction-and-the",0,"",""],["Alignment allows \"nonrobust\" decision-influences and doesn't require robust grading","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rauMEna2ddf26BqiE/alignment-allows-nonrobust-decision-influences-and-doesn-t",0,"","goodharts-law"],["Distinguishing test from training","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vCQNTuowPcnu6xqQN/distinguishing-test-from-training",0,"",""],["When to diversify? Breaking down mission-correlated investing","jh and MichaelDickens","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tgxZEei8ghtpxJoAg/when-to-diversify-breaking-down-mission-correlated-investing",0,"",""],["Why Bet Kelly?","Joe Zimmerman","2022","blog","LessWrong","www.lesswrong.com/posts/HFLuBv8NrBEysRGLZ/why-bet-kelly-1",0,"","theory"],["Why Would AI \"Aim\" To Defeat Humanity?","HoldenKarnofsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5LyKxJJfz7cYdkZfm/why-would-ai-aim-to-defeat-humanity",0,"",""],["Why Would AI \"Aim\" To Defeat Humanity?","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/why-would-ai-aim-to-defeat-humanity/",0,"",""],["Why Would AI \"Aim\" To Defeat Humanity?","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vGsRdWzwjrFgCXdMn/why-would-ai-aim-to-defeat-humanity",0,"","evals"],["Future Bowl Forecasting Tournament","ncmoulios","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yAw8afSSEFqonufPj/future-bowl-forecasting-tournament",0,"","forecasting"],["My take on Jacob Cannell’s take on AGI safety","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hsf7tQgjTZfHjiExn/my-take-on-jacob-cannell-s-take-on-agi-safety",0,"",""],["Searching for Search","NicholasKees and janus","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FDjTgDcGPc7B98AES/searching-for-search-4",0,"","interpretability"],["The Singular Value Decompositions of Transformer Weight Matrices are Highly Interpretable","beren and Sid Black","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mkbGjzxD8d8XqKHzA/the-singular-value-decompositions-of-transformer-weight",0,"","interpretability"],["Good Futures Initiative: Winter Project Internship","Aris Richardson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FZ2BMwSYhkdBWmTTA/good-futures-initiative-winter-project-internship",0,"","robustness"],["More Academic Diversity in Alignment?","ojorgensen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eKzzfLtHdG36Sr5Hw/more-academic-diversity-in-alignment",0,"",""],["Don't align agents to evaluations of plans","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fopZesxLCGAXqqaPv/don-t-align-agents-to-evaluations-of-plans",0,"","goodharts-law evals agents"],["Three Alignment Schemas & Their Problems","Shoshannah Tekofsky","2022","blog","LessWrong","www.lesswrong.com/posts/YL2RpsCsFuDBgz4HS/three-alignment-schemas-and-their-problems",0,"","forecasting"],["Fair Collective Efficient Altruism","Jobst Heitzig","2022","blog","LessWrong","www.lesswrong.com/posts/yGrL388z4WHKeerN2/fair-collective-efficient-altruism",0,"","theory"],["Mechanistic anomaly detection and ELK","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vwt3wKXWaCvqZyF74/mechanistic-anomaly-detection-and-elk",0,"","eliciting-latent-knowledge monitoring"],["Part 1: The AI Safety community has four main work groups, Strategy, Governance, Technical and Movement Building","PeterSlattery","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5iQoR8mhEpvRT43jv/part-1-the-ai-safety-community-has-four-main-work-groups",0,"","governance"],["Planes are still decades away from displacing most bird jobs","guzey","2022","blog","LessWrong","www.lesswrong.com/posts/73kwTFKgi4AagxFHJ/planes-are-still-decades-away-from-displacing-most-bird-jobs",0,"","forecasting"],["Podcast: Shoshannah Tekofsky on skilling up in AI safety, visiting Berkeley, and developing novel research ideas","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/rS4vCKLir3RphdEXh/podcast-shoshannah-tekofsky-on-skilling-up-in-ai-safety",0,"",""],["Refining the Sharp Left Turn threat model","Victoria Krakovna","2022","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2022/11/25/refining-the-sharp-left-turn-threat-model/",0,"",""],["Refining the Sharp Left Turn threat model, part 2: applying alignment techniques","Vika and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dfXwJh4X5aAcS8gF5/refining-the-sharp-left-turn-threat-model-part-2-applying",0,"","deception situational-awareness"],["Rethink Priorities’ 2022 Impact, 2023 Strategy, and Funding Gaps","kierangreig","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Liphmkodcu7XPDKfK/rethink-priorities-2022-impact-2023-strategy-and-funding-1",0,"",""],["Semi-conductor / AI stocks discussion.","sapphire","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JqBLcGYapXEG9saXD/semi-conductor-ai-stocks-discussion",0,"","governance compute-governance"],["The AI Safety community has four main work groups, Strategy, Governance, Technical and Movement Building","peterslattery","2022","blog","LessWrong","www.lesswrong.com/posts/zCYChCmnxsowBsMri/the-ai-safety-community-has-four-main-work-groups-strategy",0,"","governance"],["Clarifying wireheading terminology","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/REesy8nqvknFFKywm/clarifying-wireheading-terminology",0,"","reward-hacking"],["Corrigibility Via Thought-Process Deference","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HKZqH4QtoDcGCfcby/corrigibility-via-thought-process-deference-1",0,"",""],["Dumb and ill-posed question: Is conceptual research like this MIRI paper on the shutdown problem/Corrigibility \"real\"","joraine","2022","blog","LessWrong","www.lesswrong.com/posts/ChyQ7PgTmhfgNs8En/dumb-and-ill-posed-question-is-conceptual-research-like-this",0,"",""],["Open technical problem: A Quinean proof of Löb's theorem, for an easier cartoon guide","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rrpnEDpLPxsmmsLzs/open-technical-problem-a-quinean-proof-of-loeb-s-theorem-for",0,"",""],["Two contrasting models of “intelligence” and future growth","Magnus Vinding","2022","blog","EA Forum","forum.effectivealtruism.org/posts/7cCr6vAmN4Xi3yzR5/two-contrasting-models-of-intelligence-and-future-growth",0,"","governance forecasting"],["What I Learned Running Refine","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3zZjF3YKJ257x79mu/what-i-learned-running-refine",0,"",""],["Against a General Factor of Doom","Jeffrey Heninger","2022","blog","aiimpacts.org","aiimpacts.org/against-a-general-factor-of-doom/",0,"",""],["Announcing AI safety Mentors and Mentees","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EjX63wQoMSoCHMrmY/announcing-ai-safety-mentors-and-mentees",0,"",""],["Conjecture Second Hiring Round","Connor Leahy and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jtK7FpsqpboAfr7Td/conjecture-second-hiring-round",0,"",""],["Conjecture: a retrospective after 8 months of work","Connor Leahy and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bXTNKjsD4y3fabhwR/conjecture-a-retrospective-after-8-months-of-work-1",0,"",""],["Human-level Diplomacy was my fire alarm","Lao Mein","2022","blog","LessWrong","www.lesswrong.com/posts/AZHHEPYWvTovvtikz/human-level-diplomacy-was-my-fire-alarm",0,"","forecasting"],["Injecting some numbers into the AGI debate - by Boaz Barak","Jsevillamol","2022","blog","LessWrong","www.lesswrong.com/posts/BaQWrRgu7pjGmBByv/injecting-some-numbers-into-the-agi-debate-by-boaz-barak",0,"","forecasting"],["Notes on an Experiment with Markets","Jeffrey Heninger","2022","blog","aiimpacts.org","aiimpacts.org/notes-on-an-experiment-with-markets/",0,"",""],["Simulators, constraints, and goal agnosticism: porbynotes vol. 1","porby","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DSEwkvj8W7y8C3jau/simulators-constraints-and-goal-agnosticism-porbynotes-vol-1",0,"","chain-of-thought-faithfulness"],["What is the best source to explain short AI timelines to a skeptical person?","trevor1","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CfhMXw4hqtZshTZp3/what-is-the-best-source-to-explain-short-ai-timelines-to-a",0,"","forecasting"],["A Walkthrough of In-Context Learning and Induction Heads (w/ Charles Frye) Part 1 of 2","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xhkxv6qnnGmqmxxsz/a-walkthrough-of-in-context-learning-and-induction-heads-w",0,"","interpretability mechanistic-interpretability"],["AI will change the world, but won’t take it over by playing “3-dimensional chess”.","boazbarak and benedelman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zB3ukZJqt3pQDw9jz/ai-will-change-the-world-but-won-t-take-it-over-by-playing-3",0,"",""],["Announcing AI Alignment Awards: $100k research contests about goal misgeneralization & corrigibility","Akash and Olivia Jimenez","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JNtGxrusJRpx53Q8L/announcing-ai-alignment-awards-usd100k-research-contests",0,"",""],["Benchmarking the next generation of never-ending learners","Marc’Aurelio Ranzato and Amal Rannen-Triki","2022","blog","deepmind.com","www.deepmind.com/blog/benchmarking-the-next-generation-of-never-ending-learners",0,"","benchmarks"],["Brute-forcing the universe: a non-standard shot at diamond alignment","Martín Soto","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tAkbnojHdjqeixBiR/brute-forcing-the-universe-a-non-standard-shot-at-diamond-1",0,"",""],["Epoch is hiring a Research Data Analyst","merilalama","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5iTFKqJpSNwjk8iLv/epoch-is-hiring-a-research-data-analyst",0,"","forecasting"],["Human-level Full-Press Diplomacy (some bare facts).","Cleo Nardo","2022","blog","LessWrong","www.lesswrong.com/posts/oktnxsng7Dbc4aoZP/human-level-full-press-diplomacy-some-bare-facts",0,"","forecasting"],["imitation: Clean Imitation Learning Implementations","\\name","2022","paper","arXiv preprint","arxiv.org/abs/2211.11972",0,"","benchmarks"],["just enough spoilers for","Tamsin Leake","2022","blog","carado.moe","carado.moe/spoiler-fire-upon-deep.html",0,"",""],["Meta AI announces Cicero: Human-Level Diplomacy play (with dialogue)","Jacy Reese Anthis","2022","blog","LessWrong","www.lesswrong.com/posts/3TCYqur9YzuZ4qhtq/meta-ai-announces-cicero-human-level-diplomacy-play-with",0,"","forecasting"],["Toby Ord's new report on lessons from the development of the atomic bomb","Ishan Mukherjee","2022","blog","EA Forum","forum.effectivealtruism.org/posts/f8BY2yiLBzHLntjTL/toby-ord-s-new-report-on-lessons-from-the-development-of-the",0,"","governance"],["What is the best article to introduce someone to AI safety for the first time?","trevor1","2022","blog","EA Forum","forum.effectivealtruism.org/posts/G3JuuRsALQXLgXycL/what-is-the-best-article-to-introduce-someone-to-ai-safety",0,"",""],["[Hebbian Natural Abstractions] Introduction","Samuel Nellessen and Jan","2022","blog","LessWrong","www.lesswrong.com/posts/mFCbW6rYLzARqi5pf/hebbian-natural-abstractions-introduction",0,"",""],["Benefits/Risks of Scott Aaronson's Orthodox/Reform Framing for AI Alignment","Jeremy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hAQMQun7FySAWuQWg/benefits-risks-of-scott-aaronson-s-orthodox-reform-framing",0,"",""],["Beyond Simple Existential Risk: Survival in a Complex Interconnected World","Gideon Futerman","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cXH2sG3taM5hKbiva/beyond-simple-existential-risk-survival-in-a-complex",0,"","forecasting"],["Pre-Announcing the 2023 Open Philanthropy AI Worldviews Contest","Jason Schukraft","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3kaojgsu6qy2n8TdC/pre-announcing-the-2023-open-philanthropy-ai-worldviews",0,"",""],["Review: What We Owe The Future","Kelsey Piper","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yPpCCC4REq3zKXWdJ/review-what-we-owe-the-future",0,"",""],["ARC paper: Formalizing the presumption of independence","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wG5KTj5jFiibydgmk/arc-paper-formalizing-the-presumption-of-independence",0,"","eliciting-latent-knowledge"],["CYOAs and futurism","Tamsin Leake","2022","blog","carado.moe","carado.moe/cyoas-futurism.html",0,"",""],["Decision Theory but also Ghosts","eva_","2022","blog","LessWrong","www.lesswrong.com/posts/wjA6vAnTWxJSQKadK/decision-theory-but-also-ghosts",0,"","theory"],["let's stick with the term \"moral patient\"","Tamsin Leake","2022","blog","carado.moe","carado.moe/moral-patient-term.html",0,"",""],["A Short Dialogue on the Meaning of Reward Functions","Leon Lang and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hx48HgHzDTsSFoJui/a-short-dialogue-on-the-meaning-of-reward-functions",0,"",""],["By Default, GPTs Think In Plain Sight","Fabien Roger","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bwyKCQD7PFWKhELMr/by-default-gpts-think-in-plain-sight",0,"","interpretability"],["logical vs indexical dignity","Tamsin Leake","2022","blog","carado.moe","carado.moe/logical-indexical-dignity.html",0,"",""],["Update to Mysteries of mode collapse: text-davinci-002 not RLHF","janus","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mbGjzyy6eJXT4gFpm/update-to-mysteries-of-mode-collapse-text-davinci-002-not",0,"","rlhf"],["wonky but good enough alignment schemes","Tamsin Leake","2022","blog","carado.moe","carado.moe/wonky-good-enough-alignment.html",0,"","robustness"],["\"humans aren't aligned\" and \"human values are incoherent\"","Tamsin Leake","2022","blog","carado.moe","carado.moe/human-values-unaligned-incoherent.html",0,"",""],["Artificial Intelligence and Nuclear Command, Control, & Communications: The Risks of Integration","Peter Rautenbach","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BGFk3fZF36i7kpwWM/artificial-intelligence-and-nuclear-command-control-and-1",0,"","governance policy"],["Cognitive science and failed AI forecasts","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3nL7Ak43gmCYEFz9P/cognitive-science-and-failed-ai-forecasts",0,"","forecasting"],["Distillation of \"How Likely Is Deceptive Alignment?\"","NickGabs","2022","blog","LessWrong","www.lesswrong.com/posts/XKraEJrQRfzbCtzKN/distillation-of-how-likely-is-deceptive-alignment",0,"","alignment-faking deception"],["Don't design agents which exploit adversarial inputs","TurnTrout and Garrett Baker","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jFCK9JRLwkoJX4aJA/don-t-design-agents-which-exploit-adversarial-inputs",0,"","goodharts-law agents"],["Engineering Monosemanticity in Toy Models","Adam Jermyn and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LvznjZuygoeoTpSE6/engineering-monosemanticity-in-toy-models",0,"","interpretability"],["generalized wireheading","Tamsin Leake","2022","blog","carado.moe","carado.moe/generalized-wireheading.html",0,"","reward-hacking"],["The Disastrously Confident And Inaccurate AI","Sharat Jacob Jacob","2022","blog","LessWrong","www.lesswrong.com/posts/WNTGe87fHwDZMLqzW/the-disastrously-confident-and-inaccurate-ai",0,"",""],["Updates on scaling laws for foundation models from ' Transcending Scaling Laws with 0.1% Extra Compute'","Nick_Greig","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RDvuJaamWtT4JbsSr/updates-on-scaling-laws-for-foundation-models-from",0,"","scaling-laws"],["AI Forecasting Research Ideas","Jaime Sevilla and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ddeCNBhYc2sANsixS/ai-forecasting-research-ideas",0,"","forecasting"],["LLMs may capture key components of human agency","catubc","2022","blog","LessWrong","www.lesswrong.com/posts/ZXB3HbAuwJakwjPB6/llms-may-capture-key-components-of-human-agency",0,"","theory"],["Massive Scaling Should be Frowned Upon","harsimony","2022","blog","LessWrong","www.lesswrong.com/posts/ZqWzFDmvMZnHQZYqz/massive-scaling-should-be-frowned-upon",0,"","governance scaling-laws"],["Results from the interpretability hackathon","Esben Kran and Neel Nanda","2022","blog","LessWrong","www.lesswrong.com/posts/hhhmcWkgLwPmBuhx7/results-from-the-interpretability-hackathon",0,"","interpretability"],["The Ground Truth Problem (Or, Why Evaluating Interpretability Methods Is Hard)","Jessica Rumbelow","2022","blog","LessWrong","www.lesswrong.com/posts/snbNNQSG35D5XHtpn/the-ground-truth-problem-or-why-evaluating-interpretability",0,"","interpretability evals"],["Current themes in mechanistic interpretability research","Lee Sharkey and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Jgs7LQwmvErxR9BCC/current-themes-in-mechanistic-interpretability-research",0,"","interpretability mechanistic-interpretability"],["Disagreement with bio anchors that lead to shorter timelines","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Q3fesop6HKnemJ5Jc/disagreement-with-bio-anchors-that-lead-to-shorter-timelines",0,"","forecasting"],["Questions about Value Lock-in, Paternalism, and Empowerment","Sam","2022","blog","LessWrong","www.lesswrong.com/posts/nLjtqdhRaKcEGb4NA/questions-about-value-lock-in-paternalism-and-empowerment",0,"","power-seeking"],["Unpacking \"Shard Theory\" as Hunch, Question, Theory, and Insight","Jacy Reese Anthis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BXzhtfXeP8WwdQTXy/unpacking-shard-theory-as-hunch-question-theory-and-insight",0,"",""],["Graph of % of tasks AI is superhuman at?","Denkenberger","2022","blog","EA Forum","forum.effectivealtruism.org/posts/aCY7sjbWKYNPJDcbj/graph-of-of-tasks-ai-is-superhuman-at",0,"",""],["If FTX is liquidated, who ends up controlling Anthropic?","Ofer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qegC9AwJuWbCkj8xY/if-ftx-is-liquidated-who-ends-up-controlling-anthropic",0,"",""],["Is the speed of training large models going to increase significantly in the near future due to Cerebras Andromeda?","Amal","2022","blog","LessWrong","www.lesswrong.com/posts/vuRNYiekLSABSJPGg/is-the-speed-of-training-large-models-going-to-increase",0,"","forecasting"],["The economy as an analogy for advanced AI systems","rosehadshar and particlemania","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oH3XmScSFnZt6x2eN/the-economy-as-an-analogy-for-advanced-ai-systems-2",0,"","governance"],["The limited upside of interpretability","Peter S. Park","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TMbPEhdAAJZsSYx2L/the-limited-upside-of-interpretability",0,"","interpretability"],["Training for Good - Update & Plans for 2023","Cillian Crosson and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/22zk3tZyYWoanQwt7/training-for-good-update-and-plans-for-2023",0,"","governance robustness"],["Value Formation: An Overarching Model","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kmpNkeqEGvFue7AvA/value-formation-an-overarching-model",0,"",""],["Winners of the AI Safety Nudge Competition","Marc Carauleanu and Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pdSjwSb4GaZAApLTr/winners-of-the-ai-safety-nudge-competition",0,"",""],["AI Safety Microgrant Round","Chris Leong and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LYqkptuAiPQcmmGbs/ai-safety-microgrant-round",0,"",""],["I (with the help of a few more people) am planning to create an introduction to AI Safety that a smart teenager can understand. What am I missing?","Tapatakt","2022","blog","LessWrong","www.lesswrong.com/posts/haojehnyLfdxkgbCm/i-with-the-help-of-a-few-more-people-am-planning-to-create",0,"",""],["Two New Newcomb Variants","eva_","2022","blog","LessWrong","www.lesswrong.com/posts/Lhbkc8842L3GDDvtq/two-new-newcomb-variants",0,"","theory"],["Will we run out of ML data? Evidence from projecting dataset size trends","Pablo Villalobos","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Couhhp4pPHbbhJ2Mg/will-we-run-out-of-ml-data-evidence-from-projecting-dataset",0,"","forecasting"],["a safer experiment than quantum suicide","Tamsin Leake","2022","blog","carado.moe","carado.moe/safer-quantum-suicide-experiment.html",0,"",""],["A short critique of Vanessa Kosoy's PreDCA","Martín Soto","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FhKkFcojhKZt7nHzG/a-short-critique-of-vanessa-kosoy-s-predca-1",0,"",""],["Decision making under model ambiguity, moral uncertainty, and other agents with free will?","Jobst Heitzig","2022","blog","LessWrong","www.lesswrong.com/posts/HyFWMXJCNkpjvc9vm/decision-making-under-model-ambiguity-moral-uncertainty-and",0,"","agents theory"],["The Alignment Community Is Culturally Broken","sudo -i","2022","blog","LessWrong","www.lesswrong.com/posts/oHk9T3jbx2J5zJ39P/the-alignment-community-is-culturally-broken",0,"",""],["Will AI Worldview Prize Funding Be Replaced?","Jordan Arel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/arA65LFet5K9KDeMF/will-ai-worldview-prize-funding-be-replaced",0,"",""],["fully aligned singleton as a solution to everything","Tamsin Leake","2022","blog","carado.moe","carado.moe/fas-solution-everything.html",0,"",""],["Poster Session on AI Safety","Neil Crawford","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pcn3KDqfsxmobGazH/poster-session-on-ai-safety",0,"",""],["Vanessa Kosoy's PreDCA, distilled","Martín Soto","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EFrJdhKPZXa4MA3Gr/vanessa-kosoy-s-predca-distilled",0,"",""],["Ways to buy time","Akash and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/bkpZHXMJx3dG5waA7/ways-to-buy-time",0,"","governance"],["Apply now for the EU Tech Policy Fellowship 2023","Jan-Willem and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/adgGhehBAAjpdwKJT/apply-now-for-the-eu-tech-policy-fellowship-2023",0,"","governance policy"],["Instrumental convergence is what makes general intelligence possible","tailcalled","2022","blog","LessWrong","www.lesswrong.com/posts/GZgLa5Xc4HjwketWe/instrumental-convergence-is-what-makes-general-intelligence",0,"","instrumental-convergence"],["What are some low-cost outside-the-box ways to do/fund alignment research?","trevor1","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KKmAPEeicn5mGF93D/what-are-some-low-cost-outside-the-box-ways-to-do-fund",0,"",""],["Why I'm Working On Model Agnostic Interpretability","Jessica Rumbelow","2022","blog","LessWrong","www.lesswrong.com/posts/uXGLciramzNfb8Hvz/why-i-m-working-on-model-agnostic-interpretability",0,"","interpretability"],["Adversarial Priors: Not Paying People to Lie to You","eva_","2022","blog","LessWrong","www.lesswrong.com/posts/hedtrNfdfH3N5S5kW/adversarial-priors-not-paying-people-to-lie-to-you",0,"","theory"],["I there a demo of \"You can't fetch the coffee if you're dead\"?","Ram Rachum","2022","blog","LessWrong","www.lesswrong.com/posts/7SZq4W8eFddtkZFjH/i-there-a-demo-of-you-can-t-fetch-the-coffee-if-you-re-dead",0,"",""],["Is full self-driving an AGI-complete problem?","kraemahz","2022","blog","LessWrong","www.lesswrong.com/posts/SLckyGWZJb3bf2eCd/is-full-self-driving-an-agi-complete-problem",0,"","forecasting"],["[ASoT] Instrumental convergence is useful","Ulisse Mini","2022","blog","LessWrong","www.lesswrong.com/posts/cW3T55NeQyJnH4Px7/asot-instrumental-convergence-is-useful",0,"","instrumental-convergence"],["AI Safety groups should imitate career development clubs","Joshc","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vEAieBkRqL7Rj8KvY/ai-safety-groups-should-imitate-career-development-clubs",0,"",""],["Restricting brain organoid research to slow down AGI","freedomandutility","2022","blog","EA Forum","forum.effectivealtruism.org/posts/rvSFhWYuuBCxy5xpW/restricting-brain-organoid-research-to-slow-down-agi",0,"",""],["Trying to Make a Treacherous Mesa-Optimizer","MadHatter","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/b44zed5fBWyyQwBHL/trying-to-make-a-treacherous-mesa-optimizer",0,"","alignment-faking deception"],["A first success story for Outer Alignment: InstructGPT","Noosphere89","2022","blog","LessWrong","www.lesswrong.com/posts/sbz2sCeAuarjmvkC8/a-first-success-story-for-outer-alignment-instructgpt",0,"","rlhf"],["Applying superintelligence without collusion","Eric Drexler","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HByDKLLdaWEcA2QQD/applying-superintelligence-without-collusion",0,"","deception governance"],["Applying superintelligence without collusion","Eric Drexler","2022","blog","LessWrong","www.lesswrong.com/posts/HByDKLLdaWEcA2QQD/applying-superintelligence-without-collusion",0,"","deception governance"],["Inverse scaling can become U-shaped","Edouard Harris","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LvKmjKMvozpdmiQhP/inverse-scaling-can-become-u-shaped",0,"","scaling-laws"],["Mysteries of mode collapse","janus","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/t9svvNPNmFf5Qa3TA/mysteries-of-mode-collapse",0,"","rlhf"],["People care about each other even though they have imperfect motivational pointers?","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/heXcGuJqbx3HBmero/people-care-about-each-other-even-though-they-have-imperfect",0,"",""],["Some advice on independent research","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kpmaEevZ2KehZo2tp/some-advice-on-independent-research",0,"",""],["4 Key Assumptions in AI Safety","Prometheus","2022","blog","LessWrong","www.lesswrong.com/posts/5KAhnDbq9F4a2Y2Yg/4-key-assumptions-in-ai-safety",0,"",""],["A philosopher's critique of RLHF","ThomasW","2022","blog","LessWrong","www.lesswrong.com/posts/scnkAbvLMDjJR9WE2/a-philosopher-s-critique-of-rlhf",0,"","rlhf"],["A Walkthrough of Interpretability in the Wild (w/ authors Kevin Wang, Arthur Conmy & Alexandre Variengien)","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DZk6mRo9vhCXN9Rfn/a-walkthrough-of-interpretability-in-the-wild-w-authors",0,"","interpretability mechanistic-interpretability"],["AI Safety Unconference NeurIPS 2022","Orpheus_Lummis and Mauricio Luduena","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Z9Mprytde6BbkQcq2/ai-safety-unconference-neurips-2022",0,"",""],["Counterfactability","Scott Garrabrant","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/b2YBddoCKSixivSAJ/counterfactability",0,"",""],["How could we know that an AGI system will have good consequences?","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iDFTmb8HSGtL4zTvf/how-could-we-know-that-an-agi-system-will-have-good",0,"","robustness"],["How does one find out their AGI timelines?","Yadav","2022","blog","EA Forum","forum.effectivealtruism.org/posts/aTztN2FFRQ4GB6CRt/how-does-one-find-out-their-agi-timelines",0,"","forecasting"],["What's Happening in Australia","Bradley Tjandra and Nathan Sherburn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vGiyvfaGGFEzQsETR/what-s-happening-in-australia",0,"",""],["Examining the Differential Risk from High-level Artificial Intelligence and the Question of Control","Kyle A. Kilian and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.03157",0,"","agents forecasting"],["Has anyone increased their AGI timelines?","Darren McKee","2022","blog","LessWrong","www.lesswrong.com/posts/hr48gem2keDQvEAbg/has-anyone-increased-their-agi-timelines",0,"","forecasting"],["Longevity research as AI X-risk intervention","DirectedEvolution","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xqbm65f7TZbjfhsz4/longevity-research-as-ai-x-risk-intervention",0,"",""],["You won’t solve alignment without agent foundations","Mikhail Samin","2022","blog","LessWrong","www.lesswrong.com/posts/3dFogxGK8uNv5xCSv/you-won-t-solve-alignment-without-agent-foundations",0,"","eliciting-latent-knowledge agents theory"],["\"AGI timelines: ignore the social factor at their peril\" (Future Fund AI Worldview Prize submission)","ketanrama and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FQd2Awx8oPs9HBqev/agi-timelines-ignore-the-social-factor-at-their-peril-future",0,"","evals policy forecasting"],["\"AI predictions\" (Future Fund AI Worldview Prize submission)","ketanrama and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XxgQ9KaqDEpdxMBmc/ai-predictions-future-fund-ai-worldview-prize-submission",0,"","forecasting"],["\"Develop Anthropomorphic AGI to Save Humanity from Itself\" (Future Fund AI Worldview Prize submission)","ketanrama and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6esJGutHz9QcSuQxa/develop-anthropomorphic-agi-to-save-humanity-from-itself",0,"",""],["Instead of technical research, more people should focus on buying time","Akash and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/BbM47qBPzdSRruY4z/instead-of-technical-research-more-people-should-focus-on",0,"","governance"],["Is AI forecasting a waste of effort on the margin?","Emrik","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ctEhHxYH2a9Mrrx2f/is-ai-forecasting-a-waste-of-effort-on-the-margin",0,"","forecasting"],["My summary of “Pragmatic AI Safety”","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XxWsAw7DefKipzRLc/my-summary-of-pragmatic-ai-safety",0,"",""],["My summary of “Pragmatic AI Safety”","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/tJki2nDzxHxAax52x/my-summary-of-pragmatic-ai-safety",0,"",""],["Recommend HAIST resources for assessing the value of RLHF-related alignment research","Sam Marks and Xander Davies","2022","blog","LessWrong","www.lesswrong.com/posts/ehK7WtBsDfiCzXTw8/recommend-haist-resources-for-assessing-the-value-of-rlhf",0,"","rlhf"],["Takeaways from a survey on AI alignment resources","DanielFilan","2022","blog","LessWrong","www.lesswrong.com/posts/rXSBvSKvKdaNkhLeJ/takeaways-from-a-survey-on-ai-alignment-resources",0,"",""],["The Slippery Slope from DALLE-2 to Deepfake Anarchy","scasper","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BeGCCmDtkdJD7j5Y5/the-slippery-slope-from-dalle-2-to-deepfake-anarchy",0,"","governance"],["The Slippery Slope from DALLE-2 to Deepfake Anarchy","stecas and philljkc","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Bnp9YDqErNXHmTvvE/the-slippery-slope-from-dalle-2-to-deepfake-anarchy",0,"","governance"],["A new place to discuss cognitive science, ethics and human alignment","Daniel_Friedrich","2022","blog","EA Forum","forum.effectivealtruism.org/posts/2bfYxTt2FsGXnwDyt/a-new-place-to-discuss-cognitive-science-ethics-and-human",0,"",""],["A newcomer’s guide to the technical AI safety field","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5rsa37pBjo4Cf9fkE/a-newcomer-s-guide-to-the-technical-ai-safety-field",0,"",""],["Applications are now open for Intro to ML Safety Spring 2023","Joshc","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vHxKLNQciXN4taEdd/applications-are-now-open-for-intro-to-ml-safety-spring-2023",0,"",""],["Are alignment researchers devoting enough time to improving their research capacity?","Carson Jones","2022","blog","LessWrong","www.lesswrong.com/posts/9cKf2BBR4X2JTSeiz/are-alignment-researchers-devoting-enough-time-to-improving",0,"",""],["Don't you think RLHF solves outer alignment?","Charbel-Raphaël","2022","blog","LessWrong","www.lesswrong.com/posts/Xscch4PFgmdM3GTMY/don-t-you-think-rlhf-solves-outer-alignment",0,"","rlhf"],["For ELK truth is mostly a distraction","c.trout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NxApPkbjt9hXraSts/for-elk-truth-is-mostly-a-distraction",0,"","eliciting-latent-knowledge"],["How to store human values on a computer","oliver_siegel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FnviTNXcjG2zaYXQY/how-to-store-human-values-on-a-computer",0,"","instrumental-convergence"],["Measuring Progress on Scalable Oversight for Large Language Models","Samuel R. Bowman and 39 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.03540",0,"","scalable-oversight"],["Toy Models and Tegum Products","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3cR2YH9dpr7SmKvCb/toy-models-and-tegum-products",0,"","interpretability"],["A Mystery About High Dimensional Concept Encoding","Fabien Roger","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QHKfYy9LLAsjC5rTK/a-mystery-about-high-dimensional-concept-encoding",0,"","interpretability"],["A Theologian's Response to Anthropogenic Existential Risk","Fr Peter Wyg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EWiCySDcLSyiHTRQn/a-theologian-s-response-to-anthropogenic-existential-risk",0,"",""],["Further considerations on the Evidentialist's Wager","Martín Soto","2022","blog","LessWrong","www.lesswrong.com/posts/JxzRswbeshRmyhqTL/further-considerations-on-the-evidentialist-s-wager",0,"","theory"],["Liability regimes in the age of AI: a use-case driven analysis of the burden of proof","David Fernández Llorca and 4 others","2022","paper","Journal of Artificial Intelligence Research, Vol. 76 (2023), pp.\n  613-644","arxiv.org/abs/2211.01817",0,"","governance"],["Mechanistic Interpretability as Reverse Engineering (follow-up to \"cars and elephants\")","David Scott Krueger (formerly: capybaralet)","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kjRGMdRxXb9c5bWq5/mechanistic-interpretability-as-reverse-engineering-follow",0,"","interpretability mechanistic-interpretability"],["Why do we post our AI safety plans on the Internet?","Peter S. Park","2022","blog","LessWrong","www.lesswrong.com/posts/WercWcbpozCt4eRci/why-do-we-post-our-ai-safety-plans-on-the-internet",0,"",""],["AI Safety Needs Great Product Builders","goodgravy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pHKsedBYAvzFCniDF/ai-safety-needs-great-product-builders",0,"",""],["AI X-risk >35% mostly based on a recent peer-reviewed argument","michaelcohen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XtBJTFszs8oP3vXic/ai-x-risk-greater-than-35-mostly-based-on-a-recent-peer",0,"",""],["Announcing: What Future World? - Growing the AI Governance Community","DavidCorfield","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ppq4dZGjNHtwNd6vm/announcing-what-future-world-growing-the-ai-governance",0,"","governance"],["Humans do acausal coordination all the time","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FCffGHJnYfdE2DgRe/humans-do-acausal-coordination-all-the-time",0,"","theory"],["WFW?: Opportunity and Theory of Impact","DavidCorfield","2022","blog","EA Forum","forum.effectivealtruism.org/posts/o7e5xsGMswpK7q7ic/wfw-opportunity-and-theory-of-impact",0,"","governance policy"],["a casual intro to AI doom and alignment","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-doom.html",0,"",""],["a casual intro to AI doom and alignment","Tamsin Leake","2022","blog","LessWrong","www.lesswrong.com/posts/T4KZ62LJsxDkMf4nF/a-casual-intro-to-ai-doom-and-alignment-1",0,"",""],["Adversarial Policies Beat Professional-Level Go AIs","Tony Tong Wang","2022","paper","arXiv preprint","arxiv.org/abs/2211.00241",0,"","agents"],["AI X-Risk: Integrating on the Shoulders of Giants","TD_Pilditch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/b3nGMGGhTZawy8Zfd/ai-x-risk-integrating-on-the-shoulders-of-giants",0,"","interpretability forecasting"],["All AGI Safety questions welcome (especially basic ones) [~monthly thread]","Robert Miles","2022","blog","LessWrong","www.lesswrong.com/posts/fSMrwJnqRb5NrMYFx/all-agi-safety-questions-welcome-especially-basic-ones-2",0,"",""],["Auditing games for high-level interpretability","Paul Colognese","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EbL5W5ccwfbqFiYBJ/auditing-games-for-high-level-interpretability-1",0,"","interpretability"],["Caution when interpreting Deepmind's In-context RL paper","Sam Marks","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/avvXAvGhhGgkJDDso/caution-when-interpreting-deepmind-s-in-context-rl-paper",0,"",""],["Clarifying AI X-risk","zac_kenton and 7 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GctJD5oCDRxCspEaZ/clarifying-ai-x-risk",0,"",""],["Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","Authors: Kevin Wang and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.00593",0,"","interpretability mechanistic-interpretability evals deception"],["ML Safety Scholars Summer 2022 Retrospective","ThomasW","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pb7Q9awb5nsx3mRzk/ml-safety-scholars-summer-2022-retrospective",0,"",""],["On the correspondence between AI-misalignment and cognitive dissonance using a behavioral economics model","Stijn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LBise8JBACG9DRPG4/on-the-correspondence-between-ai-misalignment-and-cognitive",0,"",""],["Real-Time Research Recording: Can a Transformer Re-Derive Positional Info?","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/sYHrW4wwfoMBxNDcA/real-time-research-recording-can-a-transformer-re-derive",0,"","interpretability"],["Should AI focus on problem-solving or strategic planning? Why not both?","oliver_siegel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hycChZFhDQjcGcLXD/should-ai-focus-on-problem-solving-or-strategic-planning-why",0,"","governance theory"],["Threat Model Literature Review","zac_kenton and 7 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wnnkD6P2k2TfHnNmt/threat-model-literature-review",0,"",""],["\"Cars and Elephants\": a handwavy argument/analogy against mechanistic interpretability","David Scott Krueger (formerly: capybaralet)","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YEkzeJTrp69DTn8KD/cars-and-elephants-a-handwavy-argument-analogy-against",0,"","interpretability mechanistic-interpretability"],["[Book] Interpretable Machine Learning: A Guide for Making Black Box Models Explainable","Esben Kran","2022","blog","LessWrong","www.lesswrong.com/posts/x2QzeA2yAGYma4QWQ/book-interpretable-machine-learning-a-guide-for-making-black",0,"","interpretability"],["Announcing The Most Important Century Writing Prize","michel and Drew Spartz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/4XK5zkyv94voC8Fjr/announcing-the-most-important-century-writing-prize",0,"",""],["Boundaries vs Frames","Scott Garrabrant","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SZjHimszxqjNJzQWK/boundaries-vs-frames",0,"",""],["Embedding safety in ML development","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dYHiMeSdLrrX3cy4a/embedding-safety-in-ml-development",0,"",""],["My (naive) take on Risks from Learned Optimization","artkpv","2022","blog","LessWrong","www.lesswrong.com/posts/hD3zrkRm8AdfZBYtX/my-naive-take-on-risks-from-learned-optimization",0,"",""],["publishing alignment research and exfohazards","Tamsin Leake","2022","blog","carado.moe","carado.moe/publishing-infohazards.html",0,"",""],["Superintelligent AI is necessary for an amazing future, but far from sufficient","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HoQ5Rp7Gs6rebusNP/superintelligent-ai-is-necessary-for-an-amazing-future-but-1",0,"",""],["Teacher-student curriculum learning for reinforcement learning","Yanick Schraner","2022","paper","arXiv preprint","arxiv.org/abs/2210.17368",0,"","evals benchmarks deception"],["What sorts of systems can be deceptive?","Andrei Alexandru","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/atSHHCSP3NKBtqxes/what-sorts-of-systems-can-be-deceptive",0,"","alignment-faking deception"],["Instrumental ignoring AI, Dumb but not useless.","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rhiAvDqc3h29dpG34/instrumental-ignoring-ai-dumb-but-not-useless",0,"",""],["Me (Steve Byrnes) on the “Brain Inspired” podcast","Steven Byrnes","2022","blog","LessWrong","www.lesswrong.com/posts/SvWLgJj5v2E5cYP7P/me-steve-byrnes-on-the-brain-inspired-podcast",0,"",""],["«Boundaries», Part 3a: Defining boundaries as directed Markov blankets","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HrtqLy46Fx7xqRrMo/boundaries-part-3a-defining-boundaries-as-directed-markov",0,"",""],["AGI and Lock-In","Lukas Finnveden and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KqCybin8rtfP3qztq/agi-and-lock-in",0,"",""],["Is there a news-tracker about GPT-4? Why has everything become so silent about it?","Franziska Fischer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9cja9E52LCLa9Abbt/is-there-a-news-tracker-about-gpt-4-why-has-everything",0,"",""],["love, not competition","Tamsin Leake","2022","blog","carado.moe","carado.moe/love-not-competition.html",0,"",""],["aisafety.community - A living document of AI safety communities","zeshen and plex","2022","blog","LessWrong","www.lesswrong.com/posts/dEnKkYmFhXaukizWW/aisafety-community-a-living-document-of-ai-safety",0,"",""],["Join the interpretability research hackathon","Esben Kran and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vxLrFdrqRPdaHJwgs/join-the-interpretability-research-hackathon",0,"","interpretability"],["Prizes for ML Safety Benchmark Ideas","joshc","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qzXTHM7Gtxv24ew8Z/prizes-for-ml-safety-benchmark-ideas-1",0,"","benchmarks"],["Prizes for ML Safety Benchmark Ideas","Joshc and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/jo7hmLrhy576zEyiL/prizes-for-ml-safety-benchmark-ideas",0,"","benchmarks"],["Relative Behavioral Attributes: Filling the Gap between Symbolic Goal Specification and Reward Learning from Human Preferences","Lin Guan and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.15906",0,"","rlhf deception agents"],["Resources that (I think) new alignment researchers should know about","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/gcmQyyko8szuyJHyu/resources-that-i-think-new-alignment-researchers-should-know",0,"",""],["Some Lessons Learned from Studying Indirect Object Identification in GPT-2 small","KevinRoWang and 4 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3ecs6duLmTfyra3Gp/some-lessons-learned-from-studying-indirect-object",0,"","interpretability robustness"],["What should I ask Ajeya Cotra — senior researcher at Open Philanthropy, and expert on AI timelines and safety challenges?","Robert_Wiblin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/57oW7A9FpoaX76vnK/what-should-i-ask-ajeya-cotra-senior-researcher-at-open",0,"","forecasting"],["Apply to the Redwood Research Mechanistic Interpretability Experiment (REMIX), a research program in Berkeley","maxnadeau and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nqwzrpkPvviLHWXaE/apply-to-the-redwood-research-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability robustness"],["Apply to the Redwood Research Mechanistic Interpretability Experiment (REMIX), a research program in Berkeley","Max Nadeau and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MGbdhjgd2v6cg3vjv/apply-to-the-redwood-research-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability robustness"],["counterfactual computations in world models","Tamsin Leake","2022","blog","carado.moe","carado.moe/counterfactual-computation-in-world-models.html",0,"",""],["Gathering Strength, Gathering Storms: The One Hundred Year Study on Artificial Intelligence (AI100) 2021 Study Panel Report","Michael L. Littman and 16 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.15767",0,"","interpretability"],["Intent alignment should not be the goal for AGI x-risk reduction","John Nay","2022","blog","LessWrong","www.lesswrong.com/posts/Rn4wn3oqfinAsqBSf/intent-alignment-should-not-be-the-goal-for-agi-x-risk",0,"",""],["New book on s-risks","Tobias_Baumann","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XyCLLYkBCPw44jpmQ/new-book-on-s-risks",0,"",""],["Paper: In-context Reinforcement Learning with Algorithm Distillation [Deepmind]","LawrenceC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rRAfak9JRRxjsbsdk/paper-in-context-reinforcement-learning-with-algorithm",0,"",""],["Summary of \"Technology Favours Tyranny\" by Yuval Noah Harari","Madhav Malhotra","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vD3yDaDBLerMLdCQx/summary-of-technology-favours-tyranny-by-yuval-noah-harari",0,"","policy"],["Why some people believe in AGI, but I don't.","cveres","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qx6vWLwpn7joKwwAZ/why-some-people-believe-in-agi-but-i-don-t",0,"",""],["A Brief Summary Of The Most Important Century","Maynk02","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YCAEDBbskNaAc8XKx/a-brief-summary-of-the-most-important-century",0,"","governance forecasting"],["A Walkthrough of A Mathematical Framework for Transformer Circuits","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hBtjpY2wAASEpZXgN/a-walkthrough-of-a-mathematical-framework-for-transformer",0,"","interpretability mechanistic-interpretability"],["Beyond Kolmogorov and Shannon","Alexander Gietelink Oldenziel and Adam Shai","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kqxEJkq5Big9nNKxy/beyond-kolmogorov-and-shannon",0,"",""],["Maps and Blueprint; the Two Sides of the Alignment Equation","Nora_Ammann","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zAwvyBJJNu4vHWvfk/maps-and-blueprint-the-two-sides-of-the-alignment-equation",0,"",""],["Mechanism Design for AI Safety - Reading Group Curriculum","Rubi J. Hudson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ETktDQJQAR7Hgd4oS/mechanism-design-for-ai-safety-reading-group-curriculum",0,"",""],["What does it take to defend the world against out-of-control AGIs?","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LFNXiQuGrar3duBzJ/what-does-it-take-to-defend-the-world-against-out-of-control",0,"","governance"],["A Barebones Guide to Mechanistic Interpretability Prerequisites","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AaABQpuoNC8gpHf2n/a-barebones-guide-to-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability"],["Call to action: Read + Share AI Safety / Reinforcement Learning Featured in Conversation","Justin Olive","2022","blog","EA Forum","forum.effectivealtruism.org/posts/b4D3h47W58hDiHghJ/call-to-action-read-share-ai-safety-reinforcement-learning",0,"",""],["Emergent world representations: Exploring a sequence model trained on a synthetic task","Kenneth Li","2022","paper","arXiv preprint","arxiv.org/abs/2210.13382",0,"","deception"],["My (Lazy) Longtermism FAQ","Devin Kalish","2022","blog","EA Forum","forum.effectivealtruism.org/posts/SemaeDLxJe9Bsttaj/my-lazy-longtermism-faq",0,"",""],["POWERplay: An open-source toolchain to study AI power-seeking","Edouard Harris","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ojwujybfRC9SwRhAP/powerplay-an-open-source-toolchain-to-study-ai-power-seeking",0,"","instrumental-convergence power-seeking"],["The optimal timing of spending on AGI safety work; why we should probably be spending more now","Tristan Cook and Guillaume Corlouer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Ne8ZS6iJJp7EpzztP/the-optimal-timing-of-spending-on-agi-safety-work-why-we",0,"","forecasting"],["Empowerment is (almost) All We Need","jacob_cannell","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JPHeENwRyXn9YFmXc/empowerment-is-almost-all-we-need",0,"","instrumental-convergence"],["QACI: question-answer counterfactual intervals","Tamsin Leake","2022","blog","carado.moe","carado.moe/question-answer-counterfactual-intervals.html",0,"",""],["Newsletter for Alignment Research: The ML Safety Updates","Esben Kran and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ivHfucqDNeFAR5mkH/newsletter-for-alignment-research-the-ml-safety-updates",0,"",""],["Simple question about corrigibility and values in AI.","jmh","2022","blog","LessWrong","www.lesswrong.com/posts/H47Eye2LjJpxYkbDG/simple-question-about-corrigibility-and-values-in-ai",0,"",""],["Intelligent behaviour across systems, scales and substrates","Nora_Ammann","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pAXDrFTMCJtkrfREc/intelligent-behaviour-across-systems-scales-and-substrates",0,"",""],["Introducing Generally Intelligent: an AI research lab focused on improved theoretical and pragmatic understanding","joshalbrecht","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ipWNDXTdXgDfSw6fu/introducing-generally-intelligent-an-ai-research-lab-focused",0,"",""],["Learning societal values from law as part of an AGI alignment strategy","John Nay","2022","blog","LessWrong","www.lesswrong.com/posts/Tmvvvx3buP4Gj3nZK/learning-societal-values-from-law-as-part-of-an-agi",0,"","governance"],["Notes on \"Can you control the past\"","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RhAxxPXrkcEaNArnd/notes-on-can-you-control-the-past",0,"","theory"],["Scaling Laws for Reward Model Overoptimization","leogao and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/shcSdHGPhnLQkpSbX/scaling-laws-for-reward-model-overoptimization",0,"","goodharts-law scaling-laws"],["Task Phasing: Automated Curriculum Learning from Demonstrations","Vaibhav Bajaj and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.10999",0,"","deception agents policy"],["The heritability of human values: A behavior genetic critique of Shard Theory","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/bm4qeNJcc82BKJnWk/the-heritability-of-human-values-a-behavior-genetic-critique",0,"",""],["The heritability of human values: A behavior genetic critique of Shard Theory","geoffreymiller","2022","blog","LessWrong","www.lesswrong.com/posts/dRsrfC8LN4z2oehJg/the-heritability-of-human-values-a-behavior-genetic-critique",0,"",""],["Trajectories to 2036","ukc10014","2022","blog","LessWrong","www.lesswrong.com/posts/ZWRYt5FXj89AdyNf3/trajectories-to-2036",0,"","governance"],["What Does AI Alignment Success Look Like?","shminux","2022","blog","LessWrong","www.lesswrong.com/posts/ZcvNZYPsT9jvpHkp7/what-does-ai-alignment-success-look-like",0,"",""],["Governments pose larger risks than corporations: a brief response to Grace","David Johnston","2022","blog","EA Forum","forum.effectivealtruism.org/posts/w5cmtouHZxGLondEA/governments-pose-larger-risks-than-corporations-a-brief",0,"",""],["Response to Katja Grace's AI x-risk counterarguments","Erik Jenner and Johannes Treutlein","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GQat3Nrd9CStHyGaq/response-to-katja-grace-s-ai-x-risk-counterarguments",0,"",""],["Scaling laws for reward model overoptimization","OpenAI Research","2022","blog","openai.com","openai.com/research/scaling-laws-for-reward-model-overoptimization",0,"","scaling-laws"],["Should we push for requiring AI training data to be licensed?","ChristianKl","2022","blog","LessWrong","www.lesswrong.com/posts/vsuMu98Rwde5krxSJ/should-we-push-for-requiring-ai-training-data-to-be-licensed",0,"","governance training-data"],["[Link post] AI could fuel factory farming—or end it","BrianK","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cAgTyxg4azaeD6xAW/link-post-ai-could-fuel-factory-farming-or-end-it",0,"",""],["A conversation about Katja's counterarguments to AI risk","Matthew Barnett and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iXuJLARFBZbaBGxW3/a-conversation-about-katja-s-counterarguments-to-ai-risk",0,"",""],["An Extremely Opinionated Annotated List of My Favourite Mechanistic Interpretability Papers","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SfPrNY45kQaBozwmu/an-extremely-opinionated-annotated-list-of-my-favourite",0,"","interpretability mechanistic-interpretability"],["Decision theory does not imply that we get to have nice things","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rP66bz34crvDudzcJ/decision-theory-does-not-imply-that-we-get-to-have-nice",0,"","theory"],["Distilled Representations Research Agenda","Hoagy and mishajw","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wjQkQ8bgWWFym8zF9/distilled-representations-research-agenda-1",0,"","chain-of-thought-faithfulness"],["Infinite Possibility Space and the Shutdown Problem","magfrump","2022","blog","LessWrong","www.lesswrong.com/posts/F6WosiRxPHKeAk7tL/infinite-possibility-space-and-the-shutdown-problem",0,"","automated-alignment-research"],["Metaculus is building a team dedicated to AI forecasting","christian","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9dqyakpjfhuo2bmjn/metaculus-is-building-a-team-dedicated-to-ai-forecasting",0,"","forecasting"],["Science of Deep Learning - a technical agenda","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bumgqvRjTadFFkoAd/science-of-deep-learning-a-technical-agenda",0,"",""],["‘Dissolving’ AI Risk – Parameter Uncertainty in AI Future Forecasting","Froolow","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Z7r83zrSXcis6ymKo/dissolving-ai-risk-parameter-uncertainty-in-ai-future",0,"","forecasting"],["A modest case for hope","xavier rg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/juWCs6gyRvsXxPLgt/a-modest-case-for-hope",0,"",""],["A.I. Robustness: a Human-Centered Perspective on Technological Challenges and Opportunities","Andrea Tocchetti and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.08906",0,"","evals robustness"],["AI Safety Ideas: A collaborative AI safety research platform","Apart Research and Esben Kran","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DTTADonxnDRoksp4E/ai-safety-ideas-a-collaborative-ai-safety-research-platform",0,"",""],["Assistant-professor-ranked AI ethics philosopher job opportunity at Canterbury University, New Zealand","ben.smith","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Be89az6nDN37cYuri/assistant-professor-ranked-ai-ethics-philosopher-job",0,"",""],["Is interest in alignment worth mentioning for grad school applications?","Franziska Fischer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/R4nbXRipSzFECwkaE/is-interest-in-alignment-worth-mentioning-for-grad-school",0,"",""],["Maximal lotteries for value learning","ViktoriaMalyasova","2022","blog","LessWrong","www.lesswrong.com/posts/gDrSf2ccJNbbTPuG9/maximal-lotteries-for-value-learning",0,"","theory"],["Why not to solve alignment by making superintelligent humans?","Pato","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZnMZzFjJuG7kNQfnW/why-not-to-solve-alignment-by-making-superintelligent-humans",0,"",""],["Best resource to go from \"typical smart tech-savvy person\" to \"person who gets AGI risk urgency\"?","Liron","2022","blog","LessWrong","www.lesswrong.com/posts/6YpWggFWdzfCwmGqL/best-resource-to-go-from-typical-smart-tech-savvy-person-to",0,"",""],["Toward Next-Generation Artificial Intelligence: Catalyzing the NeuroAI Revolution","Anthony Zador and 26 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.08340",0,"",""],["[Job]: AI Standards Development Research Assistant","Tony Barrett","2022","blog","LessWrong","www.lesswrong.com/posts/EG3TQmwtT26gnZM2P/job-ai-standards-development-research-assistant-1",0,"","governance"],["Another problem with AI confinement: ordinary CPUs can work as radio transmitters","RomanS","2022","blog","LessWrong","www.lesswrong.com/posts/ddPu9yh65yLmMzxep/another-problem-with-ai-confinement-ordinary-cpus-can-work",0,"",""],["Counterarguments to the basic AI risk case","Katja_Grace","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zoWypGfXLmYsDFivk/counterarguments-to-the-basic-ai-risk-case",0,"",""],["Counterarguments to the basic AI x-risk case","KatjaGrace","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LDRQ5Zfqwi8GjzPYG/counterarguments-to-the-basic-ai-x-risk-case",0,"",""],["Counterarguments to the basic AI x-risk case","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/counterarguments-to-the-basic-ai-x-risk-case/",0,"",""],["Instrumental convergence: scale and physical interactions","Edouard Harris and simonsdsuo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nisaAr7wMDiMLc2so/instrumental-convergence-scale-and-physical-interactions",0,"","instrumental-convergence"],["The US expands restrictions on AI exports to China. What are the x-risk effects?","Stephen Clare","2022","blog","EA Forum","forum.effectivealtruism.org/posts/c6RnqjBd3BAkqsknB/the-us-expands-restrictions-on-ai-exports-to-china-what-are",0,"","governance compute-governance"],["The Vitalik Buterin Fellowship in AI Existential Safety is open for applications!","Cynthia Chen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wFC3axfuwABHmoQ9H/the-vitalik-buterin-fellowship-in-ai-existential-safety-is",0,"",""],["Cataloguing Priors in Theory and Practice","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EnRLAnRLG5zyJ9sAf/cataloguing-priors-in-theory-and-practice",0,"",""],["CNAS report: 'Artificial Intelligence and Arms Control'","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MBmFuoHgnow59zGfy/cnas-report-artificial-intelligence-and-arms-control",0,"","governance"],["Contra shard theory, in the context of the diamond maximizer problem","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Aet2mbnK7GDDfrEQu/contra-shard-theory-in-the-context-of-the-diamond-maximizer",0,"",""],["Greed Is the Root of This Evil","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ThtZrHooK7En9mcZr/greed-is-the-root-of-this-evil",0,"","alignment-faking deception"],["Misalignment-by-default in multi-agent systems","Edouard Harris and simonsdsuo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cemhavELfHFHRaA7Q/misalignment-by-default-in-multi-agent-systems",0,"","instrumental-convergence agents"],["ML Safety Newsletter #6","Dan Hendrycks","2022","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-6",0,"",""],["Niceness is unnatural","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/krHDNc7cDvfEL8z9a/niceness-is-unnatural",0,"",""],["Sixty years after the Cuban Missile Crisis, a new era of global catastrophic risks","christian.r","2022","blog","EA Forum","forum.effectivealtruism.org/posts/e3kLF5qPE8cRqsF8v/sixty-years-after-the-cuban-missile-crisis-a-new-era-of",0,"",""],["You are better at math (and alignment) than you think","trevor","2022","blog","LessWrong","www.lesswrong.com/posts/eyPTkNwCQoWHCdYTs/you-are-better-at-math-and-alignment-than-you-think",0,"",""],["[MLSN #6]: Transparency survey, provable robustness, ML models that predict the future","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zeAwqhjHpsJcJDmuf/mlsn-6-transparency-survey-provable-robustness-ml-models",0,"","interpretability robustness"],["A strange twist on the road to AGI","cveres","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CjifvmM3Kjn3beMyB/a-strange-twist-on-the-road-to-agi",0,"",""],["Alignment 201 curriculum","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/s4GqendqFsKzfhFzD/alignment-201-curriculum",0,"",""],["Article Review: Google's AlphaTensor","Robert_AIZI","2022","blog","LessWrong","www.lesswrong.com/posts/NcoLpvv6wS9vLCho4/article-review-google-s-alphatensor",0,"","interpretability"],["Building a transformer from scratch - AI safety up-skilling challenge","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/98jCNefEaBBb7jwu6/building-a-transformer-from-scratch-ai-safety-up-skilling",0,"",""],["Instrumental convergence in single-agent systems","Edouard Harris and simonsdsuo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pGvM95EfNXwBzjNCJ/instrumental-convergence-in-single-agent-systems",0,"","instrumental-convergence power-seeking agents"],["My argument against AGI","cveres","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZXTqMektxs2LNyMim/my-argument-against-agi",0,"",""],["[Sketch] Validity Criterion for Logical Counterfactuals","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/hkX2HWZJ7xLgRZafJ/sketch-validity-criterion-for-logical-counterfactuals",0,"","theory"],["BenevolentAI - an effectively impactful company?","Jack Hilton","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XcFk5irHSJBK2EuF3/benevolentai-an-effectively-impactful-company",0,"",""],["Human-AI Coordination via Human-Regularized Search and Learning","Hengyuan Hu and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.05125",0,"","evals benchmarks agents policy"],["Power-Seeking AI and Existential Risk","Antonio Franca","2022","blog","LessWrong","www.lesswrong.com/posts/TeSTeAwrnGtf9jwfR/power-seeking-ai-and-existential-risk",0,"","power-seeking forecasting"],["some simulation hypotheses","Tamsin Leake","2022","blog","carado.moe","carado.moe/simulation-hypotheses.html",0,"",""],["Which AI Safety Org to Join?","Yonatan Cale","2022","blog","EA Forum","forum.effectivealtruism.org/posts/RDoLDJ4toRNpMRBmk/which-ai-safety-org-to-join",0,"",""],["“Technological unemployment” AI vs. “most important century” AI: how far apart?","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZNPYMp2uu5zr3Po66/technological-unemployment-ai-vs-most-important-century-ai-1",0,"","forecasting"],["Disentangling inner alignment failures","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pWRRBtLSncELQLfrg/disentangling-inner-alignment-failures",0,"","alignment-faking deception"],["Generating Executable Action Plans with Environmentally-Aware Language Models","Maitrey Gramopadhye and Daniel Szafir","2022","paper","arXiv preprint","arxiv.org/abs/2210.04964",0,"","evals agents"],["Lessons learned from talking to >100 academics about AI safety","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SqjQFhn5KTarfW8v7/lessons-learned-from-talking-to-greater-than-100-academics",0,"",""],["outer alignment: two failure modes and past-user satisfaction","Tamsin Leake","2022","blog","carado.moe","carado.moe/outer-alignment-past-user.html",0,"",""],["QAPR 4: Inductive biases","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SxQJWw8RtXJdngBtS/qapr-4-inductive-biases",0,"",""],["When reporting AI timelines, be clear who you're deferring to","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FtggfJ2oxNSN8Niix/when-reporting-ai-timelines-be-clear-who-you-re-deferring-to",0,"","forecasting"],["AI Risk Microdynamics Survey","Froolow","2022","blog","EA Forum","forum.effectivealtruism.org/posts/8DtA57z9EyifD2wj5/ai-risk-microdynamics-survey",0,"",""],["Good ontologies induce commutative diagrams","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nLhHY2c8MWFcuWRLx/good-ontologies-induce-commutative-diagrams",0,"","robustness"],["Let’s talk about uncontrollable AI","Karl von Wendt","2022","blog","LessWrong","www.lesswrong.com/posts/6JhjHJ2rdiXcSe7tp/let-s-talk-about-uncontrollable-ai",0,"",""],["Uncontrollable AI as an Existential Risk","Karl von Wendt","2022","blog","LessWrong","www.lesswrong.com/posts/gEchYntjSXk9KXorK/uncontrollable-ai-as-an-existential-risk",0,"",""],["Don't leave your fingerprints on the future","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DJRe5obJd7kqCkvRr/don-t-leave-your-fingerprints-on-the-future",0,"",""],["Georgetown EA Fall 2022\"Intro to AI\" Reading Group","Daniel H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AsgkzmBCiFmidpndx/georgetown-ea-fall-2022-intro-to-ai-reading-group",0,"",""],["Mutual Assured Destruction used against AGI","L3opard","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TTsPA6NQY39PGYJa4/mutual-assured-destruction-used-against-agi",0,"",""],["SERI MATS Program - Winter 2022 Cohort","Ryan Kidd and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iR4kGzrWEJpXJ39ZB/seri-mats-program-winter-2022-cohort",0,"",""],["Analysis: US restricts GPU sales to China","aogara","2022","blog","LessWrong","www.lesswrong.com/posts/oBTkthd7h8sDpkiu2/analysis-us-restricts-gpu-sales-to-china",0,"","governance"],["Generating Quizzes to Support Training on Quality Management and Assurance in Space Science and Engineering","Andrés García-Silva and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.03427",0,"","evals assurance"],["Goal Misgeneralisation: Why Correct Specifications Aren’t Enough For Correct Goals","DeepMind Safety Research","2022","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/goal-misgeneralisation-why-correct-specifications-arent-enough-for-correct-goals-cf96ebc60924",0,"",""],["How undesired goals can arise with correct rewards","Rohin Shah and 3 others","2022","blog","deepmind.com","www.deepmind.com/blog/how-undesired-goals-can-arise-with-correct-rewards",0,"",""],["Knowledge-Grounded Reinforcement Learning","Zih-Yun Chiu and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.03729",0,"","interpretability agents governance policy"],["More examples of goal misgeneralization","Rohin Shah and Vikrant Varma","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cfe2LMmQC4hHTDZ8r/more-examples-of-goal-misgeneralization",0,"",""],["Polysemanticity and Capacity in Neural Networks","Buck and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kWp4R9SYgKJFHAufB/polysemanticity-and-capacity-in-neural-networks",0,"","interpretability"],["Public Explainer on AI as an Existential Risk","AndrewDoris","2022","blog","EA Forum","forum.effectivealtruism.org/posts/22xpqq5SBRGCtyXtz/public-explainer-on-ai-as-an-existential-risk-1",0,"",""],["What does it mean for an AGI to be 'safe'?","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BSee6LXg4adtrndwy/what-does-it-mean-for-an-agi-to-be-safe",0,"",""],["A shot at the diamond-alignment problem","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/k4AQqboXz8iE5TNXK/a-shot-at-the-diamond-alignment-problem",0,"",""],["AI Timelines via Cumulative Optimization Power: Less Long, More Short","jacob_cannell","2022","blog","LessWrong","www.lesswrong.com/posts/3nMpdmt8LrzxQnkGp/ai-timelines-via-cumulative-optimization-power-less-long",0,"","forecasting"],["Analysing a 2036 Takeover Scenario","ukc10014","2022","blog","LessWrong","www.lesswrong.com/posts/WAsghurJ3EppkhmQX/analysing-a-2036-takeover-scenario",0,"","governance forecasting"],["confusion about alignment requirements","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LgEvWDzWga7aagf7T/confusion-about-alignment-requirements",0,"",""],["More Recent Progress in the Theory of Neural Networks","jylin04","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nRu92PXLrdwqdtQmn/more-recent-progress-in-the-theory-of-neural-networks-1",0,"","interpretability"],["NAS-Bench-Suite-Zero: Accelerating Research on Zero Cost Proxies","Arjun Krishnakumar and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.03230",0,"","evals"],["The probability that Artificial General Intelligence will be developed by 2043 is extremely low.","cveres","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FdAfhdsSGKxP6axZY/the-probability-that-artificial-general-intelligence-will-be",0,"",""],["Warning Shots Probably Wouldn't Change The Picture Much","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/idipkijjz5PoxAwju/warning-shots-probably-wouldn-t-change-the-picture-much",0,"",""],["[Linkpost] \"Blueprint for an AI Bill of Rights\" - Office of Science and Technology Policy, USA (2022)","Fer32dwt34r3dfsz","2022","blog","LessWrong","www.lesswrong.com/posts/TAkRFJh2A3NK6oqje/linkpost-blueprint-for-an-ai-bill-of-rights-office-of",0,"","governance policy"],["confusion about alignment requirements","Tamsin Leake","2022","blog","carado.moe","carado.moe/confusion-about-alignment-requirements.html",0,"",""],["Paper: Discovering novel algorithms with AlphaTensor [Deepmind]","LawrenceC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5Zfyktwgz3rvAvZyL/paper-discovering-novel-algorithms-with-alphatensor-deepmind",0,"",""],["Reflection Mechanisms as an Alignment target: A follow-up survey","Marius Hobbhahn and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/i3pkxN43NgkLRaAGZ/reflection-mechanisms-as-an-alignment-target-a-follow-up",0,"",""],["Tracking Compute Stocks and Flows: Case Studies?","Cullen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/h3mX4esebgXnMyZSM/tracking-compute-stocks-and-flows-case-studies",0,"","governance"],["What are the risks of an oracle AI?","Griffin Young","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Ck2hHcNnvHZpFNm5T/what-are-the-risks-of-an-oracle-ai",0,"",""],["CHAI, Assistance Games, And Fully-Updated Deference [Scott Alexander]","berglund","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SbnE48y3f2Srdo4yk/chai-assistance-games-and-fully-updated-deference-scott",0,"",""],["How are you dealing with ontology identification?","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HjFAp4RiaaGqH6gNC/how-are-you-dealing-with-ontology-identification",0,"","eliciting-latent-knowledge"],["Humans aren't fitness maximizers","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CzrF5rsJWvccFdemb/humans-aren-t-fitness-maximizers",0,"",""],["Paper+Summary: OMNIGROK: GROKKING BEYOND ALGORITHMIC DATA","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jusq6kyZ6XSrtW3Bf/paper-summary-omnigrok-grokking-beyond-algorithmic-data",0,"",""],["Polysemanticity and Capacity in Neural Networks","Authors: Adam Scherlis and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.01892",0,"","interpretability evals"],["Smoke without fire is scary","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PP2Lrpvhd3bBvR8Aj/smoke-without-fire-is-scary",0,"","alignment-faking deception scaling-laws"],["A review of the Bio-Anchors report","jylin04","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TMHWfRE7zZkzgFDSo/a-review-of-the-bio-anchors-report",0,"","forecasting"],["Data for IRL: What is needed to learn human values?","Jan Wehner","2022","blog","LessWrong","www.lesswrong.com/posts/2ZKLaqKLr8TkKAxRW/data-for-irl-what-is-needed-to-learn-human-values",0,"",""],["Is there a culture overhang?","Aleksi Liimatainen","2022","blog","LessWrong","www.lesswrong.com/posts/sjcQBQvassWqGEd5F/is-there-a-culture-overhang",0,"","forecasting"],["my current outlook on AI risk mitigation","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bG7yKSRWBaMou7t93/my-current-outlook-on-ai-risk-mitigation",0,"","forecasting"],["Recall and Regurgitation in GPT2","Megan Kinniment","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qxvihKpFMuc4tvuf4/recall-and-regurgitation-in-gpt2",0,"","interpretability"],["Tony Blair Institute - Compute for AI Index ( Seeking a Supplier)","TomWestgarth","2022","blog","EA Forum","forum.effectivealtruism.org/posts/aZXgWemk6cfzYwxKB/tony-blair-institute-compute-for-ai-index-seeking-a-supplier",0,"","governance compute-governance"],["Against the weirdness heuristic","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/ZmKzbcx742mAy7xGt/against-the-weirdness-heuristic",0,"","forecasting"],["Any further work on AI Safety Success Stories?","Krieger","2022","blog","LessWrong","www.lesswrong.com/posts/TohzYjnaFr3kFaKKi/any-further-work-on-ai-safety-success-stories",0,"","governance"],["Establishing Meta-Decision-Making for AI: An Ontology of Relevance, Representation and Reasoning","Cosmin Badea and Leilani Gilpin","2022","paper","arXiv preprint","arxiv.org/abs/2210.00608",0,"","benchmarks"],["Four usages of \"loss\" in AI","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jnmG5jczvWbeRPcvG/four-usages-of-loss-in-ai",0,"","reward-hacking"],["my current outlook on AI risk mitigation","Tamsin Leake","2022","blog","carado.moe","carado.moe/outlook-ai-risk-mitigation.html",0,"",""],["Paper: Large Language Models Can Self-improve [Linkpost]","Evan R. Murphy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qwqowdhnMreKQvxLv/paper-large-language-models-can-self-improve-linkpost",0,"","chain-of-thought-faithfulness"],["Questions on databases of AI Risk estimates","Froolow","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FLTJtDmCxfpoZDA5K/questions-on-databases-of-ai-risk-estimates",0,"","forecasting"],["Why does AGI occur almost nowhere, not even just as a remark for economic/political models?","Franziska Fischer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3wcNkri9CjRC4t5Cj/why-does-agi-occur-almost-nowhere-not-even-just-as-a-remark",0,"",""],["Announcing the AI Safety Nudge Competition to Help Beat Procrastination","Marc Carauleanu and Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/c5SeLNpnHNNif6Doz/announcing-the-ai-safety-nudge-competition-to-help-beat",0,"",""],["CHAI Newsletter #2 2022","CHAI","2022","report","drive.google.com","drive.google.com/file/d/1LPIssfKeMhFVgRYRtbf19jfVws77AZvl/view?usp=sharing",0,"",""],["Do anthropic considerations undercut the evolution anchor from the Bio Anchors report?","Ege Erdil","2022","blog","LessWrong","www.lesswrong.com/posts/NHvspuLiirJwiLtfg/do-anthropic-considerations-undercut-the-evolution-anchor",0,"","forecasting"],["Google could build a conscious AI in three months","Derek Shiller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BMkDcRrGWBj2j24NB/google-could-build-a-conscious-ai-in-three-months",0,"","governance"],["(Structural) Stability of Coupled Optimizers","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zgQSsA2o6avsEsyMa/structural-stability-of-coupled-optimizers",0,"",""],["Carnegie Council MisUnderstands Longtermism","Jeff A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nTybQwrnyRMenasCc/carnegie-council-misunderstands-longtermism",0,"",""],["EAG DC: Meta-Bottlenecks in Preventing AI Doom","Joseph Bloom","2022","blog","EA Forum","forum.effectivealtruism.org/posts/F8DEipkSoTG3Zztkc/eag-dc-meta-bottlenecks-in-preventing-ai-doom",0,"",""],["Eli's review of \"Is power-seeking AI an existential risk?\"","elifland","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NLqAQzAhE9u87TvNz/eli-s-review-of-is-power-seeking-ai-an-existential-risk-1",0,"","power-seeking"],["Rethinking and Recomputing the Value of ML Models","Burcu Sayin and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.15157",0,"","evals"],["We all teach: here's how to do it better","Michael Noetel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZPNNnEu2HGNSNmifo/we-all-teach-here-s-how-to-do-it-better",0,"",""],["Builder/Breaker for Deconfusion","abramdemski","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yLTpo828duFQqPJfy/builder-breaker-for-deconfusion",0,"",""],["Clarifying the Agent-Like Structure Problem","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/moi3cFY2wpeKGu9TT/clarifying-the-agent-like-structure-problem",0,"","agents theory"],["Distribution Shifts and The Importance of AI Safety","Leon Lang","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TRKF9g65nhPBQoxJu/distribution-shifts-and-the-importance-of-ai-safety",0,"","robustness"],["FDT is not directly comparable to CDT and EDT","Sylvester Kollin","2022","blog","LessWrong","www.lesswrong.com/posts/dmjvJwCjXWE2jFbRN/fdt-is-not-directly-comparable-to-cdt-and-edt",0,"","theory"],["I'm interviewing prolific AI safety researcher Richard Ngo (now at OpenAI and previously DeepMind). What should I ask him?","Robert_Wiblin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ygdpXBoLzzsLXhhDF/i-m-interviewing-prolific-ai-safety-researcher-richard-ngo",0,"",""],["It matters when the first sharp left turn happens","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BCyK2GQKiiuYdvkst/it-matters-when-the-first-sharp-left-turn-happens",0,"","alignment-faking deception"],["Quantifying Harm","Sander Beckers and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.15111",0,"",""],["Repairing Bugs in Python Assignments Using Large Language Models","Jialu Zhang and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.14876",0,"","evals"],["Where I currently disagree with Ryan Greenblatt’s version of the ELK approach","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LBzFCPbG5s95mf43M/where-i-currently-disagree-with-ryan-greenblatt-s-version-of",0,"","eliciting-latent-knowledge"],["A Library and Tutorial for Factored Cognition with Language Models","stuhlmueller and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/X5L9g4fXmhPdQrBCA/a-library-and-tutorial-for-factored-cognition-with-language",0,"",""],["AI Safety Endgame Stories","Ivan Vendrov","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AsNjqggQQ4yJcbsWn/ai-safety-endgame-stories",0,"",""],["Estimating the Current and Future Number of AI Safety Researchers","Stephen McAleese","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3gmkrj3khJHndYGNe/estimating-the-current-and-future-number-of-ai-safety",0,"","forecasting"],["How Open Source Machine Learning Software Shapes AI","Max Langenkamp","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HsDMguLtdhFP46GQ8/how-open-source-machine-learning-software-shapes-ai",0,"","governance"],["InFi: End-to-End Learning to Filter Input for Resource-Efficiency in Mobile-Centric Inference","Mu Yuan and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.13873",0,"","evals"],["LOVE in a simbox is all you need","jacob_cannell","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WKGZBCYAbZ6WGsKHc/love-in-a-simbox-is-all-you-need",0,"",""],["Optimism, AI risk, and EA blind spots","Justis","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LjBYatyXkce5EiLDo/optimism-ai-risk-and-ea-blind-spots",0,"",""],["QAPR 3: interpretability-guided training of neural nets","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rgh4tdNrQyJYXyNs8/qapr-3-interpretability-guided-training-of-neural-nets",0,"","interpretability"],["Strange Loops - Self-Reference from Number Theory to AI","ojorgensen","2022","blog","LessWrong","www.lesswrong.com/posts/gvXAoH9gR4FSzyeCa/strange-loops-self-reference-from-number-theory-to-ai",0,"","theory"],["The missing link to AGI","Yuri Barzov","2022","blog","EA Forum","forum.effectivealtruism.org/posts/2h2E448uqCY6uGbAg/the-missing-link-to-agi",0,"",""],["Threat-Resistant Bargaining Megapost: Introducing the ROSE Value","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vJ7ggyjuP4u2yHNcP/threat-resistant-bargaining-megapost-introducing-the-rose",0,"","theory"],["Why I think strong general AI is coming soon","porby","2022","blog","LessWrong","www.lesswrong.com/posts/K4urTDkBbtNuLivJx/why-i-think-strong-general-ai-is-coming-soon",0,"","forecasting"],["7 traps that (we think) new alignment researchers often fall into","Akash and Thomas Larsen","2022","blog","LessWrong","www.lesswrong.com/posts/h5CGM5qwivGk2f5T9/7-traps-that-we-think-new-alignment-researchers-often-fall",0,"",""],["Collaborative Decision Making Using Action Suggestions","Dylan M. Asmar and Mykel J. Kochenderfer","2022","paper","arXiv preprint","arxiv.org/abs/2209.13160",0,"","agents policy"],["Failure modes in a shard theory alignment plan","Thomas Kwa","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xsieF8SXw4J5LkzEg/failure-modes-in-a-shard-theory-alignment-plan",0,"",""],["Learning When to Advise Human Decision Makers","Gali Noti and Yiling Chen","2022","paper","arXiv preprint","arxiv.org/abs/2209.13578",0,"",""],["Likelihood of an anti-AI backlash: Results from a preliminary Twitter poll","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5ZyLZjgJzyZdDLFrh/likelihood-of-an-anti-ai-backlash-results-from-a-preliminary",0,"","robustness"],["My Thoughts on the ML Safety Course","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CRMhhnKs7bymY4kbb/my-thoughts-on-the-ml-safety-course",0,"",""],["Why we're not founding a human-data-for-alignment org","LRudL and Mathieu Putz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/iBeWbfQLA9EKfsdhu/why-we-re-not-founding-a-human-data-for-alignment-org",0,"",""],["[MLSN #5]: Prize Compilation","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EmnvtFLnQBte66Ydh/mlsn-5-prize-compilation",0,"",""],["Brief Notes on Transformers","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rEPnce975Fid9v5qv/brief-notes-on-transformers",0,"",""],["existential self-determination","Tamsin Leake","2022","blog","carado.moe","carado.moe/existential-selfdet.html",0,"",""],["Inverse Scaling Prize: Round 1 Winners","Ethan Perez and Ian McKenzie","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iznohbCPFkeB9kAJL/inverse-scaling-prize-round-1-winners",0,"",""],["Lessons from Three Mile Island for AI Warning Shots","NickGabs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NyCHoZGGw5YssvDJB/lessons-from-three-mile-island-for-ai-warning-shots",0,"","governance policy"],["ML Safety Newsletter #5","Dan Hendrycks","2022","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-5",0,"",""],["Planning capacity and daemons","lukehmiles","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oNvifJbFTRebDcoHc/planning-capacity-and-daemons",0,"",""],["Project Idea: The cost of Coccidiosis on Chicken farming and if AI can help","Max Harris","2022","blog","EA Forum","forum.effectivealtruism.org/posts/uPu2tbzowqCw73dbh/project-idea-the-cost-of-coccidiosis-on-chicken-farming-and",0,"",""],["Stress Externalities More in AI Safety Pitches","NickGabs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/G6EXYYNp6KwageGZq/stress-externalities-more-in-ai-safety-pitches",0,"",""],["surprise! you want what you want","Tamsin Leake","2022","blog","carado.moe","carado.moe/surprise-you-want.html",0,"",""],["Understanding Hindsight Goal Relabeling from a Divergence Minimization Perspective","Lunjun Zhang and Bradly C. Stadie","2022","paper","arXiv preprint","arxiv.org/abs/2209.13046",0,"","agents"],["You are Underestimating The Likelihood That Convergent Instrumental Subgoals Lead to Aligned AGI","Mark Neyer","2022","blog","LessWrong","www.lesswrong.com/posts/xJ2ifnbN5PtJxtnsy/you-are-underestimating-the-likelihood-that-convergent",0,"","instrumental-convergence robustness"],["An Unexpected GPT-3 Decision in a Simple Gamble","hatta_afiq","2022","blog","LessWrong","www.lesswrong.com/posts/Gwt3gJNHc4LntD5We/an-unexpected-gpt-3-decision-in-a-simple-gamble",0,"","theory"],["AI Risk Intro 2: Solving The Problem","LRudL and TheMcDouglas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/e2upqGf6q4CiudLMu/ai-risk-intro-2-solving-the-problem",0,"",""],["Brain-over-body biases, and the embodied value problem in AI alignment","geoffreymiller","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rnkiczuRGHdgfyth3/brain-over-body-biases-and-the-embodied-value-problem-in-ai",0,"",""],["Papers to start getting into NLP-focused alignment research","Feraidoon","2022","blog","LessWrong","www.lesswrong.com/posts/utLuyMiCgLbuhsxce/papers-to-start-getting-into-nlp-focused-alignment-research",0,"","interpretability"],["Two reasons we might be closer to solving alignment than it seems","Kat Woods and Amber Dawn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/RkpdA8763yGtEovj9/two-reasons-we-might-be-closer-to-solving-alignment-than-it",0,"",""],["Two reasons we might be closer to solving alignment than it seems","KatWoods and AmberDawn","2022","blog","LessWrong","www.lesswrong.com/posts/oyZiwkxejBMuJZA7J/two-reasons-we-might-be-closer-to-solving-alignment-than-it",0,"","forecasting"],["7 Learnings and a Detailed Description of an AI Safety Reading Group","nell","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DcwdjbckGCceqctTp/7-learnings-and-a-detailed-description-of-an-ai-safety",0,"",""],["Announcing the Future Fund's AI Worldview Prize","Nick_Beckstead and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/W7C5hwq7sjdpTdrQF/announcing-the-future-fund-s-ai-worldview-prize",0,"",""],["Interlude: But Who Optimizes The Optimizer?","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/y6Wuq9ihruEAdJRvZ/interlude-but-who-optimizes-the-optimizer",0,"",""],["Interpreting Neural Networks through the Polytope Lens","Sid Black and 7 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eDicGjD9yte6FLSie/interpreting-neural-networks-through-the-polytope-lens",0,"","interpretability"],["Shahar Avin On How To Regulate Advanced AI Systems","Michaël Trazzi","2022","blog","LessWrong","www.lesswrong.com/posts/KzGwDeaYZXNWGWjd8/shahar-avin-on-how-to-regulate-advanced-ai-systems",0,"","governance"],["Shahar Avin on How to Strategically Regulate Advanced AI Systems","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/PauhAAw7Y5bHMawkT/shahar-avin-on-how-to-strategically-regulate-advanced-ai",0,"","governance"],["The Rival AI Deployment Problem: a Pre-deployment Agreement as the least-bad response","HaydnBelfield","2022","blog","EA Forum","forum.effectivealtruism.org/posts/uSH6DqjzggAYQGjxm/the-rival-ai-deployment-problem-a-pre-deployment-agreement",0,"","governance policy"],["Under what circumstances have governments cancelled AI-type systems?","David Gross","2022","blog","LessWrong","www.lesswrong.com/posts/5qAwYRhLBhvovDqft/under-what-circumstances-have-governments-cancelled-ai-type",0,"","governance"],["What are people's thoughts on working for DeepMind as a general software engineer?","Max Pietsch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JKsga96CLcxnRjzFB/what-are-people-s-thoughts-on-working-for-deepmind-as-a",0,"",""],["(My suggestions) On Beginner Steps in AI Alignment","Joseph Bloom","2022","blog","EA Forum","forum.effectivealtruism.org/posts/iyzik5qmvsQYjsqXu/my-suggestions-on-beginner-steps-in-ai-alignment",0,"",""],["[Cause Exploration Prizes] Expanding communication about AGI risks","Ines","2022","blog","EA Forum","forum.effectivealtruism.org/posts/k2tBL2nNStZEoc4tF/cause-exploration-prizes-expanding-communication-about-agi",0,"",""],["AGI Battle Royale: Why “slow takeover” scenarios devolve into a chaotic multi-AGI fight to the death","titotal","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TxrzhfRr6EXiZHv4G/agi-battle-royale-why-slow-takeover-scenarios-devolve-into-a",0,"","red-teaming forecasting"],["AI Risk Intro 2: Solving The Problem","TheMcDouglas and LRudL","2022","blog","LessWrong","www.lesswrong.com/posts/e889bGfbtbo2qrMmW/ai-risk-intro-2-solving-the-problem",0,"",""],["Crypto 'oracle protocols' for AI alignment with real-world data?","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pMoGmfg6rNsJWfZey/crypto-oracle-protocols-for-ai-alignment-with-real-world",0,"",""],["Dath Ilan's Views on Stopgap Corrigibility","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/eS7LbJizE5ucirj7a/dath-ilan-s-views-on-stopgap-corrigibility",0,"",""],["Initial Thoughts on Dissolving \"Couldness\"","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/k8a4xx25aW3jvmfF3/initial-thoughts-on-dissolving-couldness",0,"","theory"],["Mathematical Circuits in Neural Networks","Sean Osier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AQRvQ3AuQaPmuurk8/mathematical-circuits-in-neural-networks",0,"","interpretability mechanistic-interpretability"],["Methodological Therapy: An Agenda For Tackling Research Bottlenecks","adamShimi and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/foEr8gtkpzmjkvcDp/methodological-therapy-an-agenda-for-tackling-research",0,"",""],["Understanding Infra-Bayesianism: A Beginner-Friendly Video Series","Jack Parker and Connall Garrod","2022","blog","LessWrong","www.lesswrong.com/posts/mSDwPeqAzYk79vLiA/understanding-infra-bayesianism-a-beginner-friendly-video",0,"","agents theory"],["An issue with MacAskill's Evidentialist's Wager","Martín Soto","2022","blog","LessWrong","www.lesswrong.com/posts/BMkGb2ZzXdiXHaxn4/an-issue-with-macaskill-s-evidentialist-s-wager",0,"","theory"],["Announcing AISIC 2022 - the AI Safety Israel Conference, October 19-20","Davidmanheim","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/afCcihytsFtKdwSvp/announcing-aisic-2022-the-ai-safety-israel-conference",0,"",""],["EA’s brain-over-body bias, and the embodied value problem in AI alignment","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zNS53uu2tLGEJKnk9/ea-s-brain-over-body-bias-and-the-embodied-value-problem-in",0,"",""],["Establishing Oxford’s AI Safety Student Group: Lessons Learnt and Our Model","Wilkin1234 and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tnzLTnBQLEDv9zygo/establishing-oxford-s-ai-safety-student-group-lessons-learnt",0,"",""],["Introducing Whisper","OpenAI Research","2022","blog","openai.com","openai.com/research/whisper",0,"",""],["LCRL: Certified Policy Synthesis via Logically-Constrained Reinforcement Learning","Hosein Hasanbeig and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.10341",0,"","policy"],["Nearcast-based \"deployment problem\" analysis","HoldenKarnofsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vZzg8NS7wBtqcwhoJ/nearcast-based-deployment-problem-analysis",0,"",""],["Towards deconfusing wireheading and reward maximization","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jP9cKxqwqk2qQ6HiM/towards-deconfusing-wireheading-and-reward-maximization",0,"","reward-hacking"],["Toy Models of Superposition","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CTh74TaWgvRiXnkS6/toy-models-of-superposition",0,"","interpretability mechanistic-interpretability"],["Alignment Org Cheat Sheet","Akash and Thomas Larsen","2022","blog","LessWrong","www.lesswrong.com/posts/9TWReSDKyshfA66sz/alignment-org-cheat-sheet",0,"",""],["Character alignment","p.b.","2022","blog","LessWrong","www.lesswrong.com/posts/t6ZGSro4Q8fRKPont/character-alignment",0,"",""],["Doing oversight from the very start of training seems hard","peterbarnett","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/33kpQK3poHGNXJXf8/doing-oversight-from-the-very-start-of-training-seems-hard-1",0,"",""],["I'm Interviewing Kat Woods, EA Powerhouse. What Should I Ask?","SereneDesiree","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gMri6G4LajzBHgmz4/i-m-interviewing-kat-woods-ea-powerhouse-what-should-i-ask",0,"",""],["What Do AI Safety Pitches Not Get About Your Field?","Aris Richardson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hfnuwh6miJ3yn2Jpq/what-do-ai-safety-pitches-not-get-about-your-field",0,"",""],["Why AGIs utility can't outweigh humans' utility?","Alex P","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DftyvLHrfGkKgJDp9/why-agis-utility-can-t-outweigh-humans-utility",0,"",""],["PIBBSS (AI alignment) is hiring for a Project Manager","Nora_Ammann","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dm5ZL7hKti7kKw9bH/pibbss-ai-alignment-is-hiring-for-a-project-manager",0,"",""],["Quintin's alignment papers roundup - week 2","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jMRuwXdC6NPFw8HLq/quintin-s-alignment-papers-roundup-week-2",0,"",""],["Safety timelines: How long will it take to solve alignment?","Esben Kran and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9iGFjYnRquxiy29jm/safety-timelines-how-long-will-it-take-to-solve-alignment",0,"","forecasting"],["Safety timelines: How long will it take to solve alignment?","Esben Kran and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/LhEesPFocr2uT9sPA/safety-timelines-how-long-will-it-take-to-solve-alignment",0,"","forecasting"],["Summaries: Alignment Fundamentals Curriculum","Leon_Lang","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DMctFDBMcyhB7cejF/summaries-alignment-fundamentals-curriculum",0,"",""],["The ELK Framing I’ve Used","sudo -i","2022","blog","LessWrong","www.lesswrong.com/posts/SJr7accmKvz3uGLp2/the-elk-framing-i-ve-used",0,"","eliciting-latent-knowledge"],["Updates on FLI'S Value Alignment Map?","QubitSwarm99","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nKzpm3qazsXsciG29/updates-on-fli-s-value-alignment-map",0,"",""],["Aligning AI with Humans by Leveraging Legal Informatics","johnjnay","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XmKhYQfnfqb3Z7Dkr/aligning-ai-with-humans-by-leveraging-legal-informatics",0,"","governance policy"],["Inner alignment: what are we pointing at?","lukehmiles","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/E3vqfD3CLtNDNoeBr/inner-alignment-what-are-we-pointing-at",0,"",""],["Leveraging Legal Informatics to Align AI","John Nay","2022","blog","LessWrong","www.lesswrong.com/posts/9xR4KExLQKNK4iggc/leveraging-legal-informatics-to-align-ai",0,"","governance"],["Prize and fast track to alignment research at ALTER","Vanessa","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zCYGbYAaXeq7v67Km/prize-and-fast-track-to-alignment-research-at-alter",0,"",""],["Summaries: Alignment Fundamentals Curriculum","Leon Lang","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eymFwwc6jG9gPx5Zz/summaries-alignment-fundamentals-curriculum",0,"",""],["The Inter-Agent Facet of AI Alignment","Michael Oesterle","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/khr3KvExuZxdnkDtD/the-inter-agent-facet-of-ai-alignment",0,"","agents"],["A Bite Sized Introduction to ELK","Luk27182","2022","blog","LessWrong","www.lesswrong.com/posts/JjLuRtPn6B9n45Jga/a-bite-sized-introduction-to-elk",0,"","interpretability eliciting-latent-knowledge"],["Apply for mentorship in AI Safety field-building","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/sngGzPefhL5obCJue/apply-for-mentorship-in-ai-safety-field-building",0,"",""],["Prize and fast track to alignment research at ALTER","Vanessa Kosoy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8BL7w55PS4rWYmrmv/prize-and-fast-track-to-alignment-research-at-alter",0,"","agents theory"],["Refine's Third Blog Post Day/Week","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PhKSe9BT4h5peqrHL/refine-s-third-blog-post-day-week",0,"",""],["Sparse trinary weighted RNNs as a path to better language model interpretability","Am8ryllis","2022","blog","LessWrong","www.lesswrong.com/posts/Hjv5ncXk2yCKLdGbm/sparse-trinary-weighted-rnns-as-a-path-to-better-language",0,"","interpretability"],["Takeaways from our robust injury classifier project [Redwood Research]","dmz","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/n3LAgnHg6ashQK3fF/takeaways-from-our-robust-injury-classifier-project-redwood",0,"","robustness"],["[linkpost] When does technical work to reduce AGI conflict make a difference?: Introduction","antimonyanthony and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xPxkLZJ4e6yzcZWPq/linkpost-when-does-technical-work-to-reduce-agi-conflict",0,"","forecasting"],["Katja Grace on Slowing Down AI, AI Expert Surveys And Estimating AI Risk","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/2xrTTgvosGSsM85RZ/katja-grace-on-slowing-down-ai-ai-expert-surveys-and",0,"",""],["Levels of goals and alignment","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rzkCTPnkydQxfkZsX/levels-of-goals-and-alignment",0,"","alignment-faking deception"],["ordering capability thresholds","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ttRyu8u9vqX3jZFjr/ordering-capability-thresholds",0,"","forecasting"],["Refine Blogpost Day #3: The shortforms I did write","Alexander Gietelink Oldenziel","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iA9S8fLCgbjFF6fw4/refine-blogpost-day-3-the-shortforms-i-did-write",0,"",""],["Representational Tethers: Tying AI Latents To Human Ones","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/h7BA7TQTo3dxvYrek/representational-tethers-tying-ai-latents-to-human-ones",0,"","eliciting-latent-knowledge"],["The heterogeneity of human value types: Implications for AI alignment","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KZiaBCWWW3FtZXGBi/the-heterogeneity-of-human-value-types-implications-for-ai",0,"",""],["The Pugwash Conferences and the Anti-Ballistic Missile Treaty as a case study of Track II diplomacy","rani_martin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ggiCDnYcSKLxwFbBv/the-pugwash-conferences-and-the-anti-ballistic-missile",0,"","governance policy"],["The religion problem in AI alignment","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YwnfPtxHktfowyrMD/the-religion-problem-in-ai-alignment",0,"",""],["'Artificial Intelligence Governance under Change' (PhD dissertation)","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/np3KfjMadGsRc5qCm/artificial-intelligence-governance-under-change-phd",0,"","governance policy"],["Are Human Brains Universal?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/f4mbXjhQ2yaMrgLBG/are-human-brains-universal",0,"",""],["Black Box Investigations Research Hackathon","Esben Kran and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/npm6mrJQzungTLsKj/black-box-investigations-research-hackathon",0,"","interpretability"],["Capability and Agency as Cornerstones of AI risk ­— My current model","wilm","2022","blog","LessWrong","www.lesswrong.com/posts/Ezhu43CRahQSdsWug/capability-and-agency-as-cornerstones-of-ai-risk-my-current",0,"",""],["FDT defects in a realistic Twin Prisoners' Dilemma","Sylvester Kollin","2022","blog","LessWrong","www.lesswrong.com/posts/QpqKBYzPKdZpByZS3/fdt-defects-in-a-realistic-twin-prisoners-dilemma",0,"","theory"],["General advice for transitioning into Theoretical AI Safety","Martín Soto","2022","blog","LessWrong","www.lesswrong.com/posts/mEDAqbdvg6ivy7eRp/general-advice-for-transitioning-into-theoretical-ai-safety",0,"",""],["How should DeepMind's Chinchilla revise our AI forecasts?","Cleo Nardo","2022","blog","LessWrong","www.lesswrong.com/posts/kixewxJfuZ23DQDfF/how-should-deepmind-s-chinchilla-revise-our-ai-forecasts",0,"","automated-alignment-research governance forecasting scaling-laws"],["ordering capability thresholds","Tamsin Leake","2022","blog","carado.moe","carado.moe/ordering-capability-thresholds.html",0,"",""],["Why deceptive alignment matters for AGI safety","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oBFMbhQMt9HkmfF6d/why-deceptive-alignment-matters-for-agi-safety",0,"","alignment-faking deception"],["Are Speed Superintelligences Feasible for Modern ML Techniques?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/ezYSENJvqg25zwKfR/are-speed-superintelligences-feasible-for-modern-ml",0,"",""],["clippy in panpsychia","Tamsin Leake","2022","blog","carado.moe","carado.moe/clippy-in-panpsychia.html",0,"",""],["Coordinate-Free Interpretability Theory","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/sxhfSBej6gdAwcn7X/coordinate-free-interpretability-theory",0,"","interpretability"],["Emily Brontë on: Psychology Required for Serious™ AGI Safety Research","robertzk","2022","blog","LessWrong","www.lesswrong.com/posts/HhMou4Dyxj3ADcQBJ/emily-bronte-on-psychology-required-for-serious-tm-agi",0,"",""],["Forecasting thread: How does AI risk level vary based on timelines?","elifland","2022","blog","LessWrong","www.lesswrong.com/posts/h7Sx4DBL4JZnbTpes/forecasting-thread-how-does-ai-risk-level-vary-based-on",0,"","forecasting"],["Future Matters #5: supervolcanoes, AI takeover, and What We Owe the Future","Pablo and matthew.vandermerwe","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZzwMBRq5KAo6wfP4K/future-matters-5-supervolcanoes-ai-takeover-and-what-we-owe",0,"",""],["Roodman's Thoughts on Biological Anchors","lukeprog","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tAsyRARbkMym5D4jK/roodman-s-thoughts-on-biological-anchors",0,"","forecasting"],["Some ideas for epistles to the AI ethicists","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/23cMcXb2zFbfJKz3n/some-ideas-for-epistles-to-the-ai-ethicists",0,"",""],["The Defender’s Advantage of Interpretability","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6ReBeYwsDeNgv6Dr5/the-defender-s-advantage-of-interpretability",0,"","interpretability alignment-faking deception"],["When does technical work to reduce AGI conflict make a difference?: Introduction","JesseClifton and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oNQGoySbpmnH632bG/when-does-technical-work-to-reduce-agi-conflict-make-a",0,"",""],["When is intent alignment sufficient or necessary to reduce AGI conflict?","JesseClifton and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fMJhfNZXFzCNpCL8v/when-is-intent-alignment-sufficient-or-necessary-to-reduce",0,"",""],["When would AGIs engage in conflict?","JesseClifton and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cLDcKgvM6KxBhqhGq/when-would-agis-engage-in-conflict",0,"",""],["Why Do People Think Humans Are Stupid?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/yrekdsZfLsgfaFjFp/why-do-people-think-humans-are-stupid",0,"",""],["Would a Misaligned SSI Really Kill Us All?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/fyrJsfNmZECLasR3S/would-a-misaligned-ssi-really-kill-us-all",0,"",""],["Announcing an Empirical AI Safety Program","Joshc and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hkqitRNyxfzWn29AX/announcing-an-empirical-ai-safety-program",0,"",""],["Improving Language Model Prompting in Support of Semi-autonomous Task Learning","James R. Kirk and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.07636",0,"","evals agents"],["New tool for exploring EA Forum, LessWrong and Alignment Forum - Tree of Tags","Filip Sondej","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cecqH7PvsNkrxFvwe/new-tool-for-exploring-ea-forum-lesswrong-and-alignment-1",0,"",""],["Trying to find the underlying structure of computational systems","Matthias G. Mayer","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/od6zKB5swBzGL3LqE/trying-to-find-the-underlying-structure-of-computational",0,"","interpretability deception"],["[Linkpost] A survey on over 300 works about interpretability in deep networks","scasper","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WWcPFBZqpwA5kzE5y/linkpost-a-survey-on-over-300-works-about-interpretability",0,"","interpretability"],["Alignment via prosocial brain algorithms","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/yKgai84JhFmCkWQ8R/alignment-via-prosocial-brain-algorithms",0,"",""],["An experiment eliciting relative estimates for Open Philanthropy’s 2018 AI safety grants","NunoSempere","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EPhDMkovGquHtFq3h/an-experiment-eliciting-relative-estimates-for-open",0,"","evals forecasting"],["Differential technology development: preprint on the concept","Hamish_Hobbs and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/J6QCmkQmuRaP7skje/differential-technology-development-preprint-on-the-concept",0,"",""],["EA & LW Forums Weekly Summary (5 - 11 Sep 22’)","Zoe Williams","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FWcJzBdfNF3mCKP47/ea-and-lw-forums-weekly-summary-5-11-sep-22",0,"","policy"],["Ideological Inference Engines: Making Deontology Differentiable*","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FSQ4RCJobu9pussjY/ideological-inference-engines-making-deontology",0,"",""],["Resource Allocation to Agents with Restrictions: Maximizing Likelihood with Minimum Compromise","Yohai Trabelsi and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.05170",0,"","evals agents robustness"],["What could an AI-caused existential catastrophe actually look like?","Benjamin Hilton and 80000_Hours","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j3DmLmbhGQkYcZD2p/what-could-an-ai-caused-existential-catastrophe-actually",0,"",""],["Why do People Think Intelligence Will be \"Easy\"?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/nHp9dmeKrj6uQe2BF/why-do-people-think-intelligence-will-be-easy",0,"","forecasting"],["AI Risk Intro 1: Advanced AI Might Be Very Bad","LRudL and TheMcDouglas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QzrgMhTMoLe5mEas8/ai-risk-intro-1-advanced-ai-might-be-very-bad",0,"",""],["AI Risk Intro 1: Advanced AI Might Be Very Bad","TheMcDouglas and LRudL","2022","blog","LessWrong","www.lesswrong.com/posts/bJgEMfiD48fEJJxjm/ai-risk-intro-1-advanced-ai-might-be-very-bad",0,"",""],["AI Safety field-building projects I'd like to see","Akash","2022","blog","LessWrong","www.lesswrong.com/posts/DqF9c8J9LXeFFo42a/ai-safety-field-building-projects-i-d-like-to-see",0,"",""],["Briefly thinking through some analogs of debate","Eli Tyre","2022","blog","LessWrong","www.lesswrong.com/posts/pLDd8dJFq2iWuCD9h/briefly-thinking-through-some-analogs-of-debate",0,"",""],["Join ASAP (AI Safety Accountability Programme) 🚀","TheMcDouglas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pB6zRSH4Pekmh9Gmo/join-asap-ai-safety-accountability-programme",0,"",""],["Path dependence in ML inductive biases","Vivek Hebbar and evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bxkWd6WdkPqGmdHEk/path-dependence-in-ml-inductive-biases",0,"",""],["Quintin's alignment papers roundup - week 1","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/7cHgjJR2H5e4w4rxT/quintin-s-alignment-papers-roundup-week-1",0,"",""],["Unbounded utility functions and precommitment","MichaelStJules","2022","blog","LessWrong","www.lesswrong.com/posts/7jZAPw5tjyfdNG6oc/unbounded-utility-functions-and-precommitment",0,"","theory"],["A California Effect for Artificial Intelligence","henryj","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Q7gqF9ZCah2BEwZ9b/a-california-effect-for-artificial-intelligence",0,"","evals governance policy"],["Evaluations project @ ARC is hiring a researcher and a webdev/engineer","Beth Barnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/svhQMdsefdYFDq5YM/evaluations-project-arc-is-hiring-a-researcher-and-a-webdev-1",0,"","evals alignment-faking deception"],["Gatekeeper Victory: AI Box Reflection","Double and DaemonicSigil","2022","blog","LessWrong","www.lesswrong.com/posts/h5yfjFpdtqE3vATQ6/gatekeeper-victory-ai-box-reflection",0,"",""],["Markus Anderljung On The AI Policy Landscape","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Rx3baBysEhdQFzPdo/markus-anderljung-on-the-ai-policy-landscape",0,"","policy"],["Most People Start With The Same Few Bad Ideas","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Afdohjyt6gESu4ANf/most-people-start-with-the-same-few-bad-ideas",0,"",""],["Ought will host a factored cognition “Lab Meeting”","jungofthewon and stuhlmueller","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6wQyK7RaKcrkBF9nB/ought-will-host-a-factored-cognition-lab-meeting",0,"","scalable-oversight"],["Oversight Leagues: The Training Game as a Feature","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/i32eyaARtFn6eD9Cg/oversight-leagues-the-training-game-as-a-feature",0,"","goodharts-law"],["Samotsvety's AI risk forecasts","elifland and Misha_Yagudin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EG9xDM8YRz4JN4wMN/samotsvety-s-ai-risk-forecasts",0,"","forecasting"],["Swap and Scale","Stephen Fowler","2022","blog","LessWrong","www.lesswrong.com/posts/G4xCDrfpLpf9JFjKH/swap-and-scale",0,"","interpretability"],["Understanding and avoiding value drift","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jFvFreCeejRKaZv4v/understanding-and-avoiding-value-drift",0,"","deception"],["[An email with a bunch of links I sent an experienced ML researcher interested in learning about Alignment / x-safety.]","David Scott Krueger (formerly: capybaralet)","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/b2Jk3dAmerjyNDzWf/an-email-with-a-bunch-of-links-i-sent-an-experienced-ml",0,"",""],["A rough idea for solving ELK: An approach for training generalist agents like GATO to make plans and describe them to humans clearly and honestly.","Michael Soareverix","2022","blog","LessWrong","www.lesswrong.com/posts/NELtoshXv3X88kqBE/a-rough-idea-for-solving-elk-an-approach-for-training",0,"","interpretability eliciting-latent-knowledge agents"],["AI alignment with humans... but with which humans?","Geoffrey Miller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DXuwsXsqGq5GtmsB3/ai-alignment-with-humans-but-with-which-humans",0,"",""],["All AGI safety questions welcome (especially basic ones) [Sept 2022]","plex","2022","blog","LessWrong","www.lesswrong.com/posts/XpeYpKXHvbqhefQi5/all-agi-safety-questions-welcome-especially-basic-ones-sept",0,"",""],["ethics and anthropics of homomorphically encrypted computations","Tamsin Leake","2022","blog","carado.moe","carado.moe/homomorphically-encrypted-computations.html",0,"",""],["Linkpost: Github Copilot productivity experiment","Daniel Kokotajlo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/f2C4CWNmrSKMs6SaK/linkpost-github-copilot-productivity-experiment",0,"",""],["Monitoring for deceptive alignment","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Km9sHjHTsBdbgwKyi/monitoring-for-deceptive-alignment",0,"","alignment-faking deception monitoring"],["Searching for Modularity in Large Language Models","NickyP and Stephen Fowler","2022","blog","LessWrong","www.lesswrong.com/posts/rp4CiJtttvwFNHkhL/searching-for-modularity-in-large-language-models",0,"","interpretability"],["What Should AI Owe To Us? Accountable and Aligned AI Systems via Contractualist AI Alignment","xuan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cty2rSMut483QgBQ2/what-should-ai-owe-to-us-accountable-and-aligned-ai-systems",0,"","governance"],["What Should AI Owe To Us? Accountable and Aligned AI Systems via Contractualist AI Alignment","xuan","2022","blog","LessWrong","www.lesswrong.com/posts/Cty2rSMut483QgBQ2/what-should-ai-owe-to-us-accountable-and-aligned-ai-systems",0,"","governance"],["13 background claims about EA","Akash","2022","blog","EA Forum","forum.effectivealtruism.org/posts/A2YwuXe3Eo5kMZhZo/13-background-claims-about-ea",0,"",""],["AI alignment curves","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-alignment-curves.html",0,"",""],["AI-assisted list of ten concrete alignment things to do right now","lukehmiles","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3dEKykLBvCszNunzB/ai-assisted-list-of-ten-concrete-alignment-things-to-do",0,"","automated-alignment-research"],["Can \"Reward Economics\" solve AI Alignment?","Q Home","2022","blog","LessWrong","www.lesswrong.com/posts/Gdhxh45xCuKLew3bB/can-reward-economics-solve-ai-alignment",0,"","goodharts-law"],["It's (not) how you use it","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LwhzE3scZTqxERtNn/it-s-not-how-you-use-it",0,"",""],["It's (not) how you use it","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/xtgN2fJjAuziPxw84/it-s-not-how-you-use-it",0,"",""],["A New York Times article on AI risk","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fSsJXtwmzoxAy4FqG/a-new-york-times-article-on-ai-risk",0,"",""],["AI Safety Executive Summary","Sean Osier","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ixa4mM9aYF4yyqj84/ai-safety-executive-summary",0,"",""],["Alex Lawsen On Forecasting AI Progress","Michaël Trazzi","2022","blog","LessWrong","www.lesswrong.com/posts/pT86qTHDALskxCXsC/alex-lawsen-on-forecasting-ai-progress",0,"","forecasting"],["Community Building for Graduate Students: A Targeted Approach","Neil Crawford","2022","blog","LessWrong","www.lesswrong.com/posts/ZvnqqcCeSwhrmiSAe/community-building-for-graduate-students-a-targeted-approach",0,"",""],["ethics juice and anthropic juice","Tamsin Leake","2022","blog","carado.moe","carado.moe/ethic-juice-anthropic-juice.html",0,"",""],["Framing AI Childhoods","David Udell","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/saGr6DapTPKFaMhhP/framing-ai-childhoods",0,"",""],["How can we secure more research positions at our universities for x-risk researchers?","Neil Crawford","2022","blog","LessWrong","www.lesswrong.com/posts/PsdnTrwvHp95Nu2B7/how-can-we-secure-more-research-positions-at-our",0,"",""],["How Josiah became an AI safety researcher","Neil Crawford","2022","blog","LessWrong","www.lesswrong.com/posts/7GDFaqpeTThnxK5HE/how-josiah-became-an-ai-safety-researcher",0,"",""],["In conversation with AI: building better language models","Atoosa Kasirzadeh and Iason Gabriel","2022","blog","deepmind.com","www.deepmind.com/blog/in-conversation-with-ai-building-better-language-models",0,"",""],["The BLue Amazon Brain (BLAB): A Modular Architecture of Services about the Brazilian Maritime Territory","Paulo Pirozelli and 21 others","2022","paper","AI: Modeling Oceans and Climate Change (IJCAI-ECAI), 2022","arxiv.org/abs/2209.07928",0,"","agents training-data"],["A Game About AI Alignment (& Meta-Ethics): What Are the Must Haves?","JonathanErhardt","2022","blog","LessWrong","www.lesswrong.com/posts/Hk2Bp4DcdResByqm8/a-game-about-ai-alignment-and-meta-ethics-what-are-the-must",0,"",""],["AI Governance Needs Technical Work","Mauricio","2022","blog","LessWrong","www.lesswrong.com/posts/MoBQ8Y56pWLKXfcxq/ai-governance-needs-technical-work",0,"","governance"],["An entire category of risks is undervalued by EA [Summary of previous forum post]","Richard Ren","2022","blog","EA Forum","forum.effectivealtruism.org/posts/S9JeqH4qYvoLZqq9c/an-entire-category-of-risks-is-undervalued-by-ea-summary-of",0,"","governance policy"],["Beta Readers are Great","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/beta-readers-are-great/",0,"",""],["Do AI companies make their safety researchers sign a non-disparagement clause?","Ofer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wSrutNhrjMjGWWjNa/do-ai-companies-make-their-safety-researchers-sign-a-non",0,"","interpretability policy"],["Three scenarios of pseudo-alignment","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/u4E3LCPTqiqJtfhup/three-scenarios-of-pseudo-alignment",0,"",""],["Breaking Newcomb's Problem with Non-Halting states","Slimepriestess","2022","blog","LessWrong","www.lesswrong.com/posts/w5EYzCPG4AwRfc9tK/breaking-newcomb-s-problem-with-non-halting-states",0,"","theory"],["Help me find a good Hackathon subject","Charbel-Raphaël","2022","blog","LessWrong","www.lesswrong.com/posts/ND6uCdxKniFxKyBwQ/help-me-find-a-good-hackathon-subject",0,"","robustness"],["How To Know What the AI Knows - An ELK Distillation","Fabien Roger","2022","blog","LessWrong","www.lesswrong.com/posts/6ngxHbpnKYwszFqrc/how-to-know-what-the-ai-knows-an-elk-distillation",0,"","eliciting-latent-knowledge"],["program searches","Tamsin Leake","2022","blog","carado.moe","carado.moe/program-search.html",0,"",""],["The shard theory of human values","Quintin Pope and TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/iCfdcxiyr2Kj8m8mT/the-shard-theory-of-human-values",0,"",""],["An Update on Academia vs. Industry (one year into my faculty job)","David Scott Krueger (formerly: capybaralet)","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HXxHcRCxR4oHrAsEr/an-update-on-academia-vs-industry-one-year-into-my-faculty",0,"",""],["AXRP Episode 18 - Concept Extrapolation with Stuart Armstrong","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CPrJqN2Azz7Wqfiv4/axrp-episode-18-concept-extrapolation-with-stuart-armstrong",0,"",""],["Behaviour Manifolds and the Hessian of the Total Loss - Notes and Criticism","Spencer Becker-Kahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/v2SvxGNijBzRYk7Ep/behaviour-manifolds-and-the-hessian-of-the-total-loss-notes",0,"",""],["Peter Eckersley (1979-2022)","Gavin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ivep4R7LoSLhWwHGX/peter-eckersley-1979-2022",0,"",""],["Three scenarios of pseudo-alignment","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/W5nnfgWkCPxDvJMpe/three-scenarios-of-pseudo-alignment",0,"","deception"],["We may be able to see sharp left turns coming","Ethan Perez and Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2AvX8cX47CdwjbkjY/we-may-be-able-to-see-sharp-left-turns-coming",0,"",""],["Levelling Up in AI Safety Research Engineering","Gabriel Mukobi","2022","blog","LessWrong","www.lesswrong.com/posts/uLstPRyYwzfrx3enG/levelling-up-in-ai-safety-research-engineering",0,"",""],["Replacement for PONR concept","Daniel Kokotajlo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/n3w3ww9Xuf8SngBfE/replacement-for-ponr-concept",0,"","governance forecasting"],["Replacement for PONR concept","kokotajlod","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nAmauAFjjgCcDwmc6/replacement-for-ponr-concept",0,"","forecasting"],["Simulators","janus","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vJFdjigzmcXMhNTsx/simulators",0,"",""],["Simulators","janus","2022","blog","generative.ink","generative.ink/posts/simulators/",0,"",""],["Sticky goals: a concrete experiment for understanding deceptive alignment","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/a2Bxq4g2sPZwKiQmK/sticky-goals-a-concrete-experiment-for-understanding",0,"","alignment-faking deception"],["Systemic Cascading Risks: Relevance in Longtermism & Value Lock-In","Richard Ren","2022","blog","EA Forum","forum.effectivealtruism.org/posts/mWGodAi9Mv2a2EbNj/systemic-cascading-risks-relevance-in-longtermism-and-value",0,"","red-teaming governance policy"],["We Can’t Do Long Term Utilitarian Calculations Until We Know if AIs Can Be Conscious or Not","Mike20731","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Zsz3BYQTJjJdZd4DR/we-can-t-do-long-term-utilitarian-calculations-until-we-know",0,"","red-teaming"],["A Technique to Create Weaker Abstract Board Game Agents via Reinforcement Learning","Peter Jamieson and Indrima Upadhyay","2022","paper","arXiv preprint","arxiv.org/abs/2209.00711",0,"","agents"],["AI coordination needs clear wins","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vavnqwYbc8jMu3dTY/ai-coordination-needs-clear-wins",0,"",""],["AI Safety and Neighboring Communities: A Quick-Start Guide, as of Summer 2022","Sam Bowman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EFpQcBmfm2bFfM4zM/ai-safety-and-neighboring-communities-a-quick-start-guide-as",0,"",""],["Alignment is hard. Communicating that, might be harder","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/PWKWEFJMpHzFC6Qvu/alignment-is-hard-communicating-that-might-be-harder",0,"",""],["CHAI Newsletter #3 2022","CHAI","2022","report","drive.google.com","drive.google.com/file/d/1HMb90gEERyFcjf3w8WLXsI3ATNj5mnqK/view?usp=share_link",0,"",""],["Gradient Hacker Design Principles From Biology","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GdR5v7nCfKuybHHng/gradient-hacker-design-principles-from-biology",0,"",""],["I Tripped and Became GPT! (And How This Updated My Timelines)","Frankophone","2022","blog","LessWrong","www.lesswrong.com/posts/6a6tJcmKMCdJsNok2/i-tripped-and-became-gpt-and-how-this-updated-my-timelines",0,"","forecasting"],["Infra-Exercises, Part 1","Diffractor and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PRwQ6eMaEkTX2uks3/infra-exercises-part-1",0,"",""],["My take on What We Owe the Future","elifland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9Y6Y6qoAigRC7A8eX/my-take-on-what-we-owe-the-future",0,"","red-teaming forecasting"],["Reasons for my negative feelings towards the AI risk discussion","fergusq","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hLbWWuDr3EbeQqrmg/reasons-for-my-negative-feelings-towards-the-ai-risk",0,"","red-teaming"],["Strategy For Conditioning Generative Models","james.lucassen and evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HAz7apopTzozrqW2k/strategy-for-conditioning-generative-models",0,"",""],["Values lock-in is already happening (without AGI)","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ogwD28mzJy8dkwtmc/values-lock-in-is-already-happening-without-agi",0,"","red-teaming"],["A Critique of AI Takeover Scenarios","Fods12","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j7X8nQ7YvvA7Pi4BX/a-critique-of-ai-takeover-scenarios",0,"","red-teaming forecasting"],["AI Box Experiment: Are people still interested?","Double","2022","blog","LessWrong","www.lesswrong.com/posts/HrZer4yhegweXJ8RH/ai-box-experiment-are-people-still-interested",0,"",""],["From motor control to embodied intelligence","Siqi Liu and 6 others","2022","blog","deepmind.com","www.deepmind.com/blog/from-motor-control-to-embodied-intelligence",0,"",""],["Survey of NLP Researchers: NLP is contributing to AGI progress; major catastrophe plausible","Sam Bowman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3zfFPjMv9fioDAeHi/survey-of-nlp-researchers-nlp-is-contributing-to-agi",0,"",""],["The great energy descent (short version) - An important thing EA might have missed","Corentin Biteau","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wXzc75txE5hbHqYug/the-great-energy-descent-short-version-an-important-thing-ea",0,"","red-teaming"],["The great energy descent - Part 2: Limits to growth and why we probably won’t reach the stars","Corentin Biteau","2022","blog","EA Forum","forum.effectivealtruism.org/posts/8sW4h368DsoooHBNP/the-great-energy-descent-part-2-limits-to-growth-and-why-we",0,"","red-teaming"],["Chaining the evil genie: why \"outer\" AI safety is probably easy","titotal","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AoPR8BFrAFgGGN9iZ/chaining-the-evil-genie-why-outer-ai-safety-is-probably-easy",0,"","red-teaming"],["Correct-by-Construction Runtime Enforcement in AI -- A Survey","Bettina Könighofer and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.14426",0,"","agents"],["How likely is deceptive alignment?","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/A9NxPTwbw6r6Awuwt/how-likely-is-deceptive-alignment",0,"","alignment-faking deception"],["Inner Alignment via Superpowers","JamesH and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/ftw4d8kByxh39FdDR/inner-alignment-via-superpowers",0,"",""],["Outcomes of inducement prizes","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/outcomes-of-inducement-prizes/",0,"",""],["The Happiness Maximizer: Why EA is an x-risk","Obasi Shaw","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ByHc6jdXF9skwevYf/the-happiness-maximizer-why-ea-is-an-x-risk",0,"","red-teaming"],["Worlds Where Iterative Design Fails","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xFotXGEotcKouifky/worlds-where-iterative-design-fails",0,"",""],["(My understanding of) What Everyone in Technical Alignment is Doing and Why","Thomas Larsen and elifland","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QBAjndPuFbhEXKcCr/my-understanding-of-what-everyone-in-technical-alignment-is",0,"",""],["*New* Canada AI Safety & Governance community","Wyatt Tessari L'Allié","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tyyPoKWxpitEcAkw2/new-canada-ai-safety-and-governance-community",0,"","governance"],["Are Generative World Models a Mesa-Optimization Risk?","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/P6aDYBDiu9DyvsF9g/are-generative-world-models-a-mesa-optimization-risk",0,"",""],["How Do AI Timelines Affect Existential Risk?","Stephen McAleese","2022","blog","LessWrong","www.lesswrong.com/posts/cnKvxehpHqWjZJNry/how-do-ai-timelines-affect-existential-risk",0,"","forecasting"],["How might we align transformative AI if it’s developed very soon?","HoldenKarnofsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rCJQAkPTEypGjSJ8X/how-might-we-align-transformative-ai-if-it-s-developed-very",0,"","forecasting"],["How might we align transformative AI if it’s developed very soon?","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/sW6RggfddDrcmM6Aw/how-might-we-align-transformative-ai-if-it-s-developed-very",0,"","governance"],["Preventing an AI-related catastrophe - Problem profile","Benjamin Hilton and 80000_Hours","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zpReK9a8gkpGNYmBt/preventing-an-ai-related-catastrophe-problem-profile",0,"",""],["Reinforcement Learning for Hardware Security: Opportunities, Developments, and Challenges","Satwik Patnaik and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.13885",0,"","mechanistic-interpretability deception agents"],["Breaking down the training/deployment dichotomy","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/sAJnZY8pp2W3DR4mx/breaking-down-the-training-deployment-dichotomy",0,"",""],["Who ordered alignment's apple?","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hSugooaEQNTeKFsDu/who-ordered-alignment-s-apple",0,"",""],["Annual AGI Benchmarking Event","Lawrence Phillips","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/z2669YMNDt42vHugj/annual-agi-benchmarking-event",0,"","benchmarks forecasting"],["Basin broadness depends on the size and number of orthogonal features","TheMcDouglas and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EkSvsJkZE8GCeCj7u/basin-broadness-depends-on-the-size-and-number-of-orthogonal-1",0,"",""],["Help Understanding Preferences And Evil","Netcentrica","2022","blog","LessWrong","www.lesswrong.com/posts/R7TWBwiJw7gX64KEj/help-understanding-preferences-and-evil",0,"",""],["Solving Alignment by \"solving\" semantics","Q Home","2022","blog","LessWrong","www.lesswrong.com/posts/We3jtEHiBfvFzunQx/solving-alignment-by-solving-semantics",0,"","interpretability"],["The History of AI Rights Research","Jamie_Harris","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CsjKJkAyxXxmfz8su/the-history-of-ai-rights-research",0,"",""],["AGI Safety Fundamentals programme is contracting a low-code engineer","Jamie Bernardi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gsdM7hbDNrD5kpuZR/agi-safety-fundamentals-programme-is-contracting-a-low-code",0,"",""],["AI Risk in Terms of Unstable Nuclear Software","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/qPXtBGd74EBjwj6gE/ai-risk-in-terms-of-unstable-nuclear-software",0,"",""],["AI strategy nearcasting","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ktEzS3pkfeqPNh6r5/ai-strategy-nearcasting",0,"","forecasting"],["Annual AGI Benchmarking Event","Metaculus and Lawrence Phillips","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9hckjperBEnsxkzjP/annual-agi-benchmarking-event",0,"","benchmarks forecasting"],["ARIA is looking for topics for roundtables","Nathan_Barnard","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wvBsDMerZ7wrnoaLr/aria-is-looking-for-topics-for-roundtables",0,"","policy"],["DETERRENT: Detecting Trojans using Reinforcement Learning","Vasudev Gohil and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.12878",0,"","mechanistic-interpretability benchmarks agents"],["Seeking Student Submissions: Edit Your Source Code Contest","Aris Richardson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/sqsE4x2BEjK6sS2GG/seeking-student-submissions-edit-your-source-code-contest",0,"",""],["Taking the parameters which seem to matter and rotating them until they don't","Garrett Baker","2022","blog","LessWrong","www.lesswrong.com/posts/C8kn3iL9Zedorjykt/taking-the-parameters-which-seem-to-matter-and-rotating-them",0,"","interpretability"],["A Test for Language Model Consciousness","Ethan Perez","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9hxH2pxffxeeXk8YT/a-test-for-language-model-consciousness",0,"",""],["AI strategy nearcasting","HoldenKarnofsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qo2EkG3dEMv8GnX8d/ai-strategy-nearcasting",0,"","forecasting"],["Common misconceptions about OpenAI","Jacob_Hilton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3S4nyoNEEuvNsbXt8/common-misconceptions-about-openai",0,"",""],["Discovering when an agent is present in a system","DeepMind Safety Research","2022","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/discovering-when-an-agent-is-present-in-a-system-41154de11e7b",0,"","agents"],["Some conceptual alignment research projects","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/27AWRKbKyXuzQoaSk/some-conceptual-alignment-research-projects",0,"",""],["The Shard Theory Alignment Scheme","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/ZNXDRGshgoq3cmxhB/the-shard-theory-alignment-scheme",0,"","interpretability"],["Who would you have on your dream team for solving AGI Alignment?","Greg_Colbourn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AHoZX4JwFS2ygAoZR/who-would-you-have-on-your-dream-team-for-solving-agi",0,"",""],["Your posts should be on arXiv","JanBrauner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TYTEJxzeK3jBMq2TZ/your-posts-should-be-on-arxiv",0,"",""],["AI Safety For Dummies (Like Me)","Madhav Malhotra","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pEjBEJHAoNuqS4pWH/ai-safety-for-dummies-like-me",0,"",""],["Beliefs and Disagreements about Automating Alignment Research","Ian McKenzie","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JKgGvJCzNoBQss2bq/beliefs-and-disagreements-about-automating-alignment",0,"","automated-alignment-research"],["Ethan Perez on the Inverse Scaling Prize, Language Feedback and Red Teaming","Michaël Trazzi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xPrtsLAyMcZkkszvw/ethan-perez-on-the-inverse-scaling-prize-language-feedback",0,"","red-teaming"],["Google AI integrates PaLM with robotics: SayCan update [Linkpost]","Evan R. Murphy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CuBKm8bkfWhYegcw8/google-ai-integrates-palm-with-robotics-saycan-update",0,"",""],["Interspecies diplomacy as a potentially productive lens on AGI alignment","Shariq Hashme","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JhYcvwWCECmEXtxkj/interspecies-diplomacy-as-a-potentially-productive-lens-on",0,"",""],["Should I force myself to work on AGI alignment?","Isaac Benson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kCu2cANxdkr7ferQ4/should-i-force-myself-to-work-on-agi-alignment",0,"",""],["Thoughts about OOD alignment","Catnee","2022","blog","LessWrong","www.lesswrong.com/posts/7kcDTJmEPAnSbyeYh/thoughts-about-ood-alignment",0,"","robustness"],["Vingean Agency","abramdemski","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fWiYdyicEaCHCcAKx/vingean-agency",0,"",""],["What Makes A Good Measurement Device?","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/aKoeYyHZTRu8izMwA/what-makes-a-good-measurement-device",0,"","interpretability robustness"],["AGI Timelines Are Mostly Not Strategically Relevant To Alignment","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/d7RFGPMkbKurQjW6o/agi-timelines-are-mostly-not-strategically-relevant-to",0,"","forecasting"],["AI alignment as “navigating the space of intelligent behaviour”","Nora_Ammann","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FuToH2KHxKmJLGk2B/ai-alignment-as-navigating-the-space-of-intelligent",0,"",""],["First call for EA Data Science/ML/AI","astrastefania","2022","blog","EA Forum","forum.effectivealtruism.org/posts/GfutYj4PCTavF9fcx/first-call-for-ea-data-science-ml-ai",0,"",""],["Philanthropists Probably Shouldn't Mission-Hedge AI Progress","MichaelDickens","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JD6QvQG3q5p6heKuA/philanthropists-probably-shouldn-t-mission-hedge-ai-progress",0,"",""],["Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","Deep Ganguli and 25 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.07858",0,"","rlhf interpretability red-teaming scaling-laws"],["The Brussels Effect and Artificial Intelligence: How EU regulation will impact the global AI market","Charlotte Siegmann and Markus Anderljung","2022","paper","arXiv preprint","arxiv.org/abs/2208.12645",0,"","governance"],["Artificial Intelligence: A Modern Approach, Chapters 1-17","Stuart Russell and Peter Norvig","2022","report","aima.cs.berkeley.edu","aima.cs.berkeley.edu/",0,"",""],["Finding Goals in the World Model","Jeremy Gillen and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jnyTqPRHwcieAXgrA/finding-goals-in-the-world-model",0,"",""],["Information-Theoretic Equivalence of Entropic Multi-Marginal Optimal Transport: A Theory for Multi-Agent Communication","Shuchan Wang","2022","paper","arXiv preprint","arxiv.org/abs/2208.10256",0,"","agents"],["What if we solve AI Safety but no one cares","142857","2022","blog","LessWrong","www.lesswrong.com/posts/4rgDink5LirgzwyqF/what-if-we-solve-ai-safety-but-no-one-cares",0,"","governance"],["AXRP Episode 17 - Training for Very High Reliability with Daniel Ziegler","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3hJhAsZHXP44Ljah5/axrp-episode-17-training-for-very-high-reliability-with",0,"","robustness"],["My Plan to Build Aligned Superintelligence","apollonianblues","2022","blog","LessWrong","www.lesswrong.com/posts/EaDN9grmTfzXbM652/my-plan-to-build-aligned-superintelligence",0,"",""],["Pivotal acts using an unaligned AGI?","Simon Fischer","2022","blog","LessWrong","www.lesswrong.com/posts/FsEDu6CvzyzJkrQd8/pivotal-acts-using-an-unaligned-agi",0,"",""],["Reward Reports for Reinforcement Learning","Thomas Krendl Gilbert and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.10817",0,"","robustness"],["Benchmarking Proposals on Risk Scenarios","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DsYe7TKc4NhyJuPEy/benchmarking-proposals-on-risk-scenarios",0,"","benchmarks"],["Broad Picture of Human Values","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kQKcEkzKmpPe6qxDb/broad-picture-of-human-values-1",0,"",""],["everything is okay","Tamsin Leake","2022","blog","carado.moe","carado.moe/everything-is-okay.html",0,"",""],["Less Threat-Dependent Bargaining Solutions?? (3/2)","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WvHpsqmwQopnaMh46/less-threat-dependent-bargaining-solutions-3-2",0,"","theory"],["No One-Size-Fit-All Epistemic Strategy","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/du92yeHQn9iE5vorj/no-one-size-fit-all-epistemic-strategy",0,"",""],["PreDCA: vanessa kosoy's alignment protocol","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WcWzLSn8ZjJhCZxP4/predca-vanessa-kosoy-s-alignment-protocol",0,"",""],["Reducing Goodhart: Announcement, Executive Summary","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2drtQcFoyRCjYaJFe/reducing-goodhart-announcement-executive-summary",0,"","goodharts-law"],["Refine's Second Blog Post Day","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EXkY7drHc6jdNxCmB/refine-s-second-blog-post-day",0,"",""],["What if we approach AI safety like a technical engineering safety problem","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zNYmbFwgrxiNtayMm/what-if-we-approach-ai-safety-like-a-technical-engineering",0,"",""],["BERI, Epoch, and FAR will explain their work & current job openings online this Sunday","Rockwell","2022","blog","EA Forum","forum.effectivealtruism.org/posts/G334oF8rmwWEAyhQD/beri-epoch-and-far-will-explain-their-work-and-current-job",0,"",""],["Compute & Antitrust: Regulatory implications of the AI hardware supply chain, from chip design to cloud APIs","HaydnBelfield and Shin-ShinHua","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CQxd84nNgojYwLucb/compute-and-antitrust-regulatory-implications-of-the-ai",0,"","governance policy compute-governance"],["Effective Enforceability of EU Competition Law Under Different AI Development Scenarios: A Framework for Legal Analysis","HaydnBelfield and Shin-ShinHua","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tLCnuSG6jZnm33FB9/effective-enforceability-of-eu-competition-law-under",0,"","governance policy"],["Epistemic Artefacts of (conceptual) AI alignment research","Nora_Ammann and particlemania","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CewHdaAjEvG3bpc6C/epistemic-artefacts-of-conceptual-ai-alignment-research",0,"",""],["How to do theoretical research, a personal perspective","Mark Xu","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zbFGRRWPwxBwjwknY/how-to-do-theoretical-research-a-personal-perspective-1",0,"",""],["PreDCA: vanessa kosoy's alignment protocol","Tamsin Leake","2022","blog","carado.moe","carado.moe/predca.html",0,"",""],["Alignment's phlogiston","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DCtqgsywCRakLvHn6/alignment-s-phlogiston",0,"",""],["Announcing Encultured AI: Building a Video Game","Andrew_Critch and Nick Hay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ALkH4o53ofm862vxc/announcing-encultured-ai-building-a-video-game",0,"",""],["Discovering Agents","zac_kenton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XxX2CAoFskuQNkBDy/discovering-agents",0,"","agents theory"],["Discovering when an agent is present in a system","Zachary Kenton and 5 others","2022","blog","deepmind.com","www.deepmind.com/blog/discovering-when-an-agent-is-present-in-a-system",0,"","agents"],["Intellectual Property Evaluation Utilizing Machine Learning","Jinxin Ding and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.08611",0,"","evals"],["An Exercise in Speed-Reading: The National Security Commission on AI (NSCAI) Final Report","abiolvera","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xmun77hGeBbg4AjxJ/an-exercise-in-speed-reading-the-national-security",0,"",""],["Autonomy as taking responsibility for reference maintenance","Ramana Kumar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/KH8fcM6SK8EkcKcZr/autonomy-as-taking-responsibility-for-reference-maintenance",0,"",""],["Concrete Advice for Forming Inside Views on AI Safety","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LbSWMfbAHQnFpApQ7/concrete-advice-for-forming-inside-views-on-ai-safety",0,"",""],["Conditioning, Prompts, and Fine-Tuning","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/chevXfQmRYrTZnj8r/conditioning-prompts-and-fine-tuning",0,"",""],["Could realistic depictions of catastrophic AI risks effectively reduce said risks?","Matthew Barber","2022","blog","EA Forum","forum.effectivealtruism.org/posts/evoMzZWPbkGmPeJvB/could-realistic-depictions-of-catastrophic-ai-risks",0,"",""],["Discovering Agents","Zachary Kenton and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.08345",0,"","agents policy"],["Human Mimicry Mainly Works When We’re Already Close","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TLvbXNHNvppNEkXYj/human-mimicry-mainly-works-when-we-re-already-close",0,"","automated-alignment-research"],["Interpretability Tools Are an Attack Channel","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/rytFP2zRYNK85rFyX/interpretability-tools-are-an-attack-channel",0,"","interpretability"],["Matt Yglesias on AI Policy","Grant Demaree","2022","blog","LessWrong","www.lesswrong.com/posts/SyNQ6LaTuntWpaQJu/matt-yglesias-on-ai-policy",0,"","governance policy"],["Mesa-optimization for goals defined only within a training environment is dangerous","Rubi J. Hudson","2022","blog","LessWrong","www.lesswrong.com/posts/RBsSXWL6QuDsoSMJ6/mesa-optimization-for-goals-defined-only-within-a-training",0,"",""],["The Core of the Alignment Problem is...","Thomas Larsen and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vMM6HmSQaKmKadvBi/the-core-of-the-alignment-problem-is-1",0,"",""],["The longest training run","Jsevillamol and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RihYwmskuJT9Rkbjq/the-longest-training-run",0,"",""],["A concern about the “evolutionary anchor” of Ajeya Cotra’s report on AI timelines.","NunoSempere","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FHTyixYNnGaQfEexH/a-concern-about-the-evolutionary-anchor-of-ajeya-cotra-s",0,"","red-teaming forecasting"],["A Review of the Convergence of 5G/6G Architecture and Deep Learning","Olusola T. Odeyomi and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.07643",0,"",""],["alignment research is very weird","Tamsin Leake","2022","blog","carado.moe","carado.moe/alignment-research-is-very-weird.html",0,"",""],["alignment researchspace is potentially malign","Tamsin Leake","2022","blog","carado.moe","carado.moe/alignment-researchspace-is-malign.html",0,"",""],["carmack predictions","Tamsin Leake","2022","blog","carado.moe","carado.moe/carmack-predictions.html",0,"",""],["Deception as the optimal: mesa-optimizers and inner alignment","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Rvmu5LLz8qGnFcGLz/deception-as-the-optimal-mesa-optimizers-and-inner-alignment",0,"","deception"],["Deception as the optimal: mesa-optimizers and inner alignment","Eleni Angelou","2022","blog","LessWrong","www.lesswrong.com/posts/u256AQr2xiNAgPftG/deception-as-the-optimal-mesa-optimizers-and-inner-alignment",0,"","deception"],["guiding your brain: go with your gut!","Tamsin Leake","2022","blog","carado.moe","carado.moe/go-with-your-gut.html",0,"",""],["Supplement to \"The Brussels Effect and AI: How EU AI regulation will impact the global AI market\"","MarkusAnderljung and Charlotte","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gJGMFdGqFhs3mKo2s/supplement-to-the-brussels-effect-and-ai-how-eu-ai",0,"","evals governance policy"],["The Credibility of Apocalyptic Claims: A Critique of Techno-Futurism within Existential Risk","Ember","2022","blog","EA Forum","forum.effectivealtruism.org/posts/a2XaDeadFe6eHfDwG/the-credibility-of-apocalyptic-claims-a-critique-of-techno",0,"","red-teaming"],["What Makes an Idea Understandable? On Architecturally and Culturally Natural Ideas.","NickyP and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/bQkp4pi5Ra4SgSSxn/what-makes-an-idea-understandable-on-architecturally-and",0,"","interpretability"],["A Mechanistic Interpretability Analysis of Grokking","Neel Nanda and Tom Lieberum","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/N6WM6hs7RQMKDhYjB/a-mechanistic-interpretability-analysis-of-grokking",0,"","interpretability mechanistic-interpretability"],["Seeking Interns/RAs for Mechanistic Interpretability Projects","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kx9K7tLFf8rxcnNLT/seeking-interns-ras-for-mechanistic-interpretability",0,"","interpretability mechanistic-interpretability"],["The Parable of the Boy Who Cried 5% Chance of Wolf","Kat Woods","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5BpgZKFrfeRtREg7W/the-parable-of-the-boy-who-cried-5-chance-of-wolf",0,"",""],["What's General-Purpose Search, And Why Might We Expect To See It In Trained ML Systems?","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6mysMAqvo9giHC4iX/what-s-general-purpose-search-and-why-might-we-expect-to-see",0,"",""],["A brief note on Simplicity Bias","Spencer Becker-Kahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gyggp2DJRMRLSnhid/a-brief-note-on-simplicity-bias-1",0,"",""],["A general framework for reward function distances.","E Jenner and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=de&user=8DgF8HcAAAAJ&sortby=pubdate&citation_for_view=8DgF8HcAAAAJ:IjCSPb-OGe4C",0,"",""],["A neural network ensemble with feature engineering for improved credit card fraud detection.","E Esenogho and 4 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:M3ejUd6NZC8C",0,"",""],["A Penalty Default Approach to Preemptive Harm Disclosure and Mitigation for AI Systems.","RJ Yew and D Hadfield-Menell","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:maZDTaKrznsC",0,"",""],["A Primer on Maximum Causal Entropy Inverse Reinforcement Learning.","Adam Gleave and Sam Toyer","2022","report","gleave.me","www.gleave.me/publication/2022-03-mce-irl-primer/",0,"",""],["A Unified Survey on Anomaly, Novelty, Open-Set, and Out-of-Distribution Detection: Solutions and Future Challenges.","Mohammadreza Salehi and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2110.14051",0,"","robustness monitoring"],["Actionable Guidance for High-Consequence AI Risk Management: Towards Standards Addressing AI Catastrophic Risks.","Anthony M and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.08966",0,"",""],["Active uncertainty learning for human-robot interaction: An implicit dual control approach.","H Hu and JF Fisac","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:-f6ydRqryjwC",0,"",""],["Active uncertainty reduction for human-robot interaction: An implicit dual control approach.","H Hu and JF Fisac","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:TQgYirikUcIC",0,"",""],["AdaCat: Adaptive Categorical Discretization for Autoregressive Models.","Qiyang (Colin) Li and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2208.02246",0,"",""],["Adversarial Motion Priors Make Good Substitutes for Complex Reward Functions.","Alejandro Escontrela and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.15103",0,"","agents robustness"],["Adversarial Policies Beat Professional-Level Go AIs.","Tony Wang and 9 others","2022","report","gleave.me","www.gleave.me/publication/2022-11-go-attack/",0,"",""],["AI experts are increasingly afraid of what they’re creating.","Stuart Russell","2022","report","vox.com","www.vox.com/the-highlight/23447596/artificial-intelligence-agi-openai-gpt3-existential-risk-human-extinction",0,"",""],["All the posts I will never write","Alexander Gietelink Oldenziel","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EeTq9vbzMT4Zb4oWo/all-the-posts-i-will-never-write",0,"",""],["An empirical investigation of representation learning for imitation.","Xin Chen and 11 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=9pawpdMAAAAJ&sortby=pubdate&citation_for_view=9pawpdMAAAAJ:2osOgNQ5qMEC",0,"",""],["An interpretable machine learning approach for hepatitis b diagnosis.","George Obaido and 12 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:hC7cP41nSMkC",0,"","interpretability"],["APReL: A Library for Active Preference-based Reward Learning Algorithms.","Erdem Bıyık and 2 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:qUcmZB5y_30C",0,"",""],["Are we living in an AGI World?.","Stuart Russell","2022","report","share.transistor.fm","share.transistor.fm/s/deca2e46",0,"",""],["ASHA: Assistive Teleoperation via Human-in-the-Loop Reinforcement Learning. .","S and 12 others","2022","report","sites.google.com","sites.google.com/view/asha-assist",0,"",""],["Assistive Teaching of Motor Control Tasks to Humans.","M Srivastava and 4 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:HDshCWvjkbEC",0,"",""],["Automatic Correction of Human Translations.","Jessy Lin and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.08593",0,"",""],["Autoregressive Latent Video Prediction with High-Fidelity Image Generator.","Younggyo Seo and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.07143",0,"","benchmarks"],["Autoregressive Uncertainty Modeling for 3D Bounding Box Prediction.","YuXuan (Andrew) Liu and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.07424",0,"",""],["Back to the Future: Efficient, Time-Consistent Solutions in Reach-Avoid Games.","DR Anthony and 3 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:L8Ckcad2t8MC",0,"",""],["Balancing Efficiency and Comfort in Robot-Assisted Bite Transfer.","Suneel Belkhale and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2111.11401",0,"","evals robustness"],["Banning Lethal Autonomous Weapons: An Education.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/papers/issues22-laws.pdf",0,"",""],["Better Conflict Bulletin.","J Stray","2022","report","betterconflictbulletin.substack.com","betterconflictbulletin.substack.com/",0,"",""],["Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.","Aarohi Srivastava and 39 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:HDshCWvjkbEC",0,"",""],["Brain computation as fast spiking neural Monte Carlo inference in probabilistic programs.","George Matheos and 6 others","2022","report","george.matheos.com","george.matheos.com/publication/22-snmc/",0,"",""],["Brain-like AGI project \"aintelope\"","Gunnar_Zarncke","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/c2tEfqEMi6jcJ4kdg/brain-like-agi-project-aintelope",0,"",""],["Building Human Values into Recommender Systems: An Interdisciplinary Synthesis.","Jonathan Stray and 20 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.10192",0,"","policy"],["Calculus on MDPs: Potential Shaping as a Gradient.","Erik Jenner and 2 others","2022","report","gleave.me","www.gleave.me/publication/2022-08-calculus-mdps/",0,"",""],["Chain of Thought Imitation with Procedure Cloning.","Mengjiao (Sherry) Yang and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.10816",0,"","deception agents chain-of-thought-faithfulness"],["Choices, Risks, and Reward Reports: Charting Public Policy for Reinforcement Learning Systems.","Thomas Krendl Gilbert and 3 others","2022","report","cltc.berkeley.edu","cltc.berkeley.edu/reward-reports/",0,"","policy"],["COLA: Consistent Learning with Opponent-Learning Awareness.","Timon Willi* and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.04098",0,"","evals agents"],["Conditional Imitation Learning for Multi-Agent Games.","Andy Shih and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.01448",0,"","evals agents policy"],["Cooperative and uncooperative institution designs: Surprises and problems in open-source game theory.","A Critch and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=F3_yOXUAAAAJ&cstart=20&pagesize=80&citation_for_view=F3_yOXUAAAAJ:dhFuZR0502QC",0,"",""],["Cooperative Multi-Agent Fairness and Equivariant Policies.","NA Grupen and 2 others","2022","report","ojs.aaai.org","ojs.aaai.org/index.php/AAAI/article/view/21166",0,"","agents"],["COVID-19 diagnosis: a review of rapid antigen, RT-PCR and artificial intelligence methods.","Raphael Taiwo Aruleba and 7 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:dhFuZR0502QC",0,"",""],["Cross-Domain Imitation Learning via Optimal Transport.","Arnaud Fickinger and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2110.03684",0,"","agents"],["DayDreamer: World Models for Physical Robot Learning.","Philipp Wu and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.14176",0,"",""],["Deciding to be authentic: Intuition is favored over deliberation when authenticity matters.","K Oktar and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:SpbeaW3--B0C",0,"",""],["Democratic Control of Recommender Systems.","J Stray","2022","report","archive.org","archive.org/details/stray-metagov-20221123\n\nhttps://docs.google.com/presentation/d/1Q8Qccpr2n3vqA2f-7xDvrf838NJXNMYeWRnot5HPcro/edit?usp=sharing",0,"",""],["Demography of Machine Learning Education Within the K12.","GO K Aruleba and 2 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:qxL8FJ1GzNcC",0,"",""],["Diagnostics for Deep Neural Networks with Automated Copy/Paste Attacks.","S Casper and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:ns9cj8rnVeAC",0,"",""],["Differential Assessment of Black-Box AI Agents..","Rashmeet Kaur Nayyar* and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.13236",0,"","evals agents policy"],["Director: Deep Hierarchical Planning from Pixels.","Danijar Hafner and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.04114",0,"","interpretability agents policy"],["Discovered policy optimisation.","C Lu and 5 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=wMjQdBcAAAAJ&sortby=pubdate&citation_for_view=wMjQdBcAAAAJ:Tyk-4Ss8FVUC",0,"","policy"],["Disentangling Abstraction from Statistical Pattern Matching in Human and Machine Learning..","Kumar and 14 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.01437",0,"","evals deception agents"],["Eliciting Compatible Demonstrations for Multi-Human Imitation Learning.","Kanishk Gandhi and 3 others","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/gandhi2022eliciting.pdf",0,"",""],["Enhanced prediction of chronic kidney disease using feature selection and boosted classifiers.","ID Mienye and 3 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:7PzlFSSx8tAC",0,"",""],["essential inequality vs functional inequivalence","Tamsin Leake","2022","blog","carado.moe","carado.moe/essential-inequality-vs-functional-inequivalence.html",0,"",""],["Estimating and Penalizing Induced Preference Shifts in Recommender Systems.","Micah Carroll and 3 others","2022","paper","Proceedings of the 39th International Conference on Machine\n  Learning, PMLR 162:2686-2708, 2022","arxiv.org/abs/2204.11966",0,"","evals"],["Ethical Explanations.","C Lewry and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:XoXfffV-tXoC",0,"",""],["Evaluations of Causal Claims Reflect a Trade-Off Between Informativeness and Compression.","D Kinney and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:HbR8gkJAVGIC",0,"","evals"],["Experiments on causal exclusion.","T Blanchard and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:ILKRHgRFtOwC",0,"",""],["Explaining Reinforcement Learning Policies through Counterfactual Trajectories.","J Frost and 6 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Rd18rbYAAAAJ&sortby=pubdate&citation_for_view=Rd18rbYAAAAJ:qjMakFHDy7sC",0,"",""],["Explanations and Causal Judgments Are Differentially Sensitive to Covariation and Mechanism Information.","N Vasil and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:0N-VGjzr574C",0,"",""],["Exploiting Extensive-Form Structure in Empirical Game-Theoretic Analysis.","C KONICKI and 2 others","2022","report","strategicreasoning.org","strategicreasoning.org/exploiting-extensive-form-structure-in-empirical-game-theoretic-analysis/",0,"",""],["Few-Shot Preference Learning for Human-in-the-Loop RL.","Donald Joseph Hejna III and Dorsa Sadigh","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/hejna2022fewshot.pdf",0,"",""],["First Contact: Unsupervised Human-Machine Co-Adaptation via Mutual Information Maximization. .","S and 6 others","2022","report","sites.google.com","sites.google.com/view/coadaptation?pli=1",0,"",""],["Fleet-DAgger: Interactive Robot Fleet Learning with Scalable Human Supervision.","Ryan Hoque and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.14349",0,"","evals benchmarks policy"],["For learning in symmetric teams, local optima are global nash equilibria.","S Emmons and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=F3_yOXUAAAAJ&cstart=20&pagesize=80&citation_for_view=F3_yOXUAAAAJ:9ZlFYXVOiuMC",0,"",""],["Get It in Writing: Formal Contracts Mitigate Social Dilemmas in Multi-Agent RL.","PJK Christoffersen and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:NMxIlDl6LWMC",0,"","agents"],["Goal Misgeneralization: Why Correct Specifications Aren’t Enough For Correct Goals.","R Shah and 6 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=odFQXSYAAAAJ&sortby=pubdate&citation_for_view=odFQXSYAAAAJ:MXK_kJrjxJIC",0,"",""],["Graph Value Iteration.","D Feng and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.09608",0,"","deception"],["Guided imitation of task and motion planning.","MJ McDonald and D Hadfield-Menell","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:R3hNpaxXUhUC",0,"",""],["Haptic perception using optoelectronic robotic flesh for embodied artificially intelligent agents.","Jose A Barreiros and 8 others","2022","report","science.org","www.science.org/doi/abs/10.1126/scirobotics.abi6745",0,"","agents"],["Heterogeneous-agent mirror learning: A continuum of solutions to cooperative marl.","JG Kuba and 5 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=wMjQdBcAAAAJ&sortby=pubdate&citation_for_view=wMjQdBcAAAAJ:Y0pCki6q_DkC",0,"","agents"],["Hidden Gold for IT Professionals, Educators, and Students: Insights From Stack Overflow Survey.","OA Dada and 4 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:9ZlFYXVOiuMC",0,"",""],["Hierarchical Few-Shot Imitation with Skill Transition Models.","Kourosh Hakhamaneshi and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2107.08981",0,"","agents"],["How do people incorporate advice from artificial agents when making physical judgments?.","E Brockbank and 6 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:ZeXyd9-uunAC",0,"","agents"],["How to talk so AI will learn: Instructions, descriptions, and autonomy.","T Sumers and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:RGFaLdJalmkC",0,"",""],["How to talk so your robot will learn: Instructions, descriptions, and pragmatics.","TR Sumers and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:TFP_iSt0sucC",0,"",""],["How Would The Viewer Feel? Estimating Wellbeing From Video Scenarios.","Mantas Mazeika and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.10039",0,"",""],["How “is” shapes “ought” for folk-biological concepts.","E Foster-Hanson and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=20&pagesize=80&citation_for_view=Z4CpYzsAAAAJ:M7yex6snE4oC",0,"",""],["Human-Centered Evaluation of Explanations.","Jordan Boyd-Graber and 6 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:ClCfbGk0d_YC",0,"","evals"],["If we succeed.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/future/#:~:text=Stuart%20Russell%2C-,If%20we%20succeed,-.%20Daedalus%2C%20Spring",0,"",""],["Imitation Learning by Estimating Expertise of Demonstrators.","Mark Beliaev* and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.01288",0,"","policy"],["imitation: Clean Imitation Learning Implementations.","Adam Gleave and 10 others","2022","report","gleave.me","www.gleave.me/publication/2022-11-imitation/",0,"",""],["Inferring Rewards from Language in Context.","Daniel Fried and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.02515",0,"","deception"],["Information acquisition under resource limitations in a noisy environment.","J Halpern and 2 others","2022","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/abstract.html#inattention",0,"",""],["Information Technology Roles and Their Most-Used Programming Languages.","OA Dada and 4 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:aqlVkmm33-oC",0,"",""],["Instruction-Following Agents with Jointly Pre-Trained Vision-Language Models.","Hao Liu and 3 others","2022","report","google.com","www.google.com/url?q=https%3A%2F%2Fopenreview.net%2Fforum%3Fid%3DReNyLYfUdr&sa=D&sntz=1&usg=AOvVaw0tyQpIena-NhduVgz_U7Ot",0,"","agents"],["Invariance in Policy Optimisation and Partial Identifiability in Reward Learning.","Joar Skalse and 4 others","2022","report","gleave.me","www.gleave.me/publication/2022-03-invariance-policy/",0,"","policy"],["Is there any research or forecasts of how likely AI Alignment is going to be a hard vs. easy problem relative to capabilities?","Jordan Arel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gKiaFWDt99Eatm82k/is-there-any-research-or-forecasts-of-how-likely-ai",0,"","forecasting"],["ISAACS: Iterative Soft Adversarial Actor-Critic for Safety.","KC Hsu and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:R3hNpaxXUhUC",0,"",""],["Israel’s Autonomous Urban Quadcopter Brings ‘Search & Attack In One’.","Stuart Russell","2022","report","forbes.com","www.forbes.com/sites/davidhambling/2022/11/11/israels-urban-quadcopter-brings-search--attack-in-one",0,"",""],["It Takes Four to Tango: Multiagent Self Play for Automatic Curriculum Generation.","Yuqing Du and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.10608",0,"","agents policy"],["JEDAI: A System for Skill-Aligned Explainable Robot Planning.","Naman Shah and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2111.00585",0,"","deception"],["Joint Communication and Motion Planning for Cobots.","Mehdi Dadvar and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2109.14004",0,"","evals agents"],["Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents.","Wenlong Huang and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.07207",0,"","evals agents"],["Learning Bimanual Scooping Policies for Food Acquisition.","Jennifer Grannen* and 3 others","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/grannen2022learning.pdf",0,"","robustness"],["Learning Deterministic Finite Automata Decompositions from Examples and Demonstrations.","N and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.13013",0,"","interpretability"],["Learning from Humans for Adaptive Interaction.","E Bıyık","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:dhFuZR0502QC",0,"",""],["Learning from Imperfect Demonstrations via Adversarial Confidence Transfer.","Zhangjie Cao* and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.02967",0,"","policy"],["Learning Latent Actions to Control Assistive Robots.","Dylan Losey and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2107.02907",0,"","evals"],["Learning multimodal rewards from rankings.","V Myers and 3 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&citation_for_view=P-G3sjYAAAAJ:Wp0gIr-vW9MC",0,"",""],["Learning Multimodal Rewards from Rankings.","Vivek Myers and 3 others","2022","report","proceedings.mlr.press","proceedings.mlr.press/v164/myers22a.html",0,"",""],["Learning Preferences for Interactive Autonomy.","E Bıyık","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:mB3voiENLucC",0,"",""],["Learning Representations that Enable Generalization in Assistive Tasks.","J and 13 others","2022","paper","arXiv preprint","arxiv.org/abs/2212.03175",0,"","evals deception robustness"],["Learning reward functions from diverse sources of human feedback: Optimally integrating demonstrations and preferences.","E Bıyık and 5 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&citation_for_view=P-G3sjYAAAAJ:8k81kl-MbHgC",0,"","rlhf"],["Learning Visual Robotic Control Efficiently with Contrastive Pre-training and Data Augmentation.","Albert Zhan and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2012.07975",0,"",""],["Learning Visuo-Haptic Skewering Strategies for Robot-Assisted Feeding.","Priya Sundaresan and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.14648",0,"","policy robustness"],["Left Heavy Tails and the Effectiveness of the Policy and Value Networks in DNN-based best-first search for Sokoban Planning.","D Feng and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=pJ28HA0AAAAJ&cstart=200&pagesize=100&citation_for_view=pJ28HA0AAAAJ:Dem6FJhTUoYC",0,"","policy"],["Lethal Autonomous Weapons.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/media/DW_2022_Autonomous_Weapons.mp4",0,"",""],["Leveraging Smooth Attention Prior for Multi-Agent Trajectory Prediction.","Z Cao and 3 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:7PzlFSSx8tAC",0,"","agents"],["Linguistic communication as (inverse) reward design.","TR Sumers and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:iH-uZ7U-co4C",0,"",""],["Masked Autoencoding for Scalable and Generalizable Decision Making.","Fangchen Liu and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.12740",0,"","agents scaling-laws"],["Masked World Models for Visual Control.","Younggyo Seo and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.14244",0,"",""],["Mechanisms of Belief Persistence in the Face of Societal Disagreement.","K Oktar and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:foquWX3nUaYC",0,"",""],["Mens Rea in Moral Judgment and Criminal Law.","C Giffin and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:e_rmSamDkqQC",0,"",""],["Metareasoning for Safe Decision Making in Autonomous Systems.","Justin Svegliato and 3 others","2022","report","justinsvegliato.com","justinsvegliato.com/pdf/SBSZicra22.pdf",0,"","agents"],["Microdrones: the AI assassins set to become weapons of mass destruction.","Stuart Russell","2022","report","telegraph.co.uk","www.telegraph.co.uk/global-health/terror-and-security/drone-assassins-micro-killing-machine/",0,"",""],["Mining Frequently Traveled Routes During COVID-19.","G Obaido and 3 others","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:QIV2ME_5wuYC",0,"",""],["Mirror learning: A unifying framework of policy optimisation.","J Grudzien and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=wMjQdBcAAAAJ&sortby=pubdate&citation_for_view=wMjQdBcAAAAJ:YsMSGLbcyi4C",0,"","policy"],["Motivated to learn: An account of explanatory satisfaction.","EG Liquin and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=20&pagesize=80&citation_for_view=Z4CpYzsAAAAJ:oNZyr7d5Mn4C",0,"",""],["Multi-agent reinforcement learning is a sequence modeling problem.","M Wen and 6 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=wMjQdBcAAAAJ&sortby=pubdate&citation_for_view=wMjQdBcAAAAJ:zYLM7Y9cAGgC",0,"","agents"],["Multi-Objective Policy Gradients with Topological Constraints.","Kyle Wray* and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2209.07096",0,"","policy"],["No‑regret Learning in Dynamic Stackelberg Games.","N and 9 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.04786",0,"","deception policy"],["Object representations as fixed points: Training iterative refinement algorithms with implicit differentiation.","S&C\nChang and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.00787",0,"",""],["On testing for discrimination using causal models.","J Halpern and 2 others","2022","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/abstract.html#discrimination",0,"",""],["On the computational consequences of cost function design in nonlinear optimal control.","T Westenbroek and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=qYXPDjQAAAAJ&citation_for_view=qYXPDjQAAAAJ:9yKSN-GCB0IC",0,"",""],["On the Effectiveness of Fine-tuning Versus Meta-reinforcement Learning.","Zhao Mandi and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.03271",0,"","evals benchmarks agents"],["Optimal Behavior Prior: Improving Human-AI Collaboration Through Generalizable Human Models..","Mesut Yang and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.01602",0,"","agents robustness"],["Optimal conservative offline RL with general function approximation via augmented Lagrangian.","Paria Rashidinejad and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=BgQkdsYAAAAJ&sortby=pubdate&citation_for_view=BgQkdsYAAAAJ:W7OEmFMy1HYC",0,"",""],["Optimal nudging for cognitively bounded agents: A framework for modeling, predicting, and controlling the effects of choice architectures.","Callaway and 5 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/publications.php#:~:text=of%20choice%20architectures.-,(preprint%20link),-RPM",0,"","agents"],["Pairwise Weights for Temporal Credit Assignment.","Zeyu Zheng and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2102.04999",0,"","policy"],["PantheonRL: A MARL Library for Dynamic Training Interactions.","Bidipta Sarkar* and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2112.07013",0,"","agents"],["Partner-Aware Algorithms in Decentralized Cooperative Bandit Teams.","E Biyik and 4 others","2022","report","scholar.google.com.tr","scholar.google.com.tr/citations?view_op=view_citation&hl=en&user=P-G3sjYAAAAJ&cstart=20&pagesize=80&citation_for_view=P-G3sjYAAAAJ:hC7cP41nSMkC",0,"",""],["Path Independent Equilibrium Models Can Better Exploit Test-Time Computation.","Cem Anil* and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.09961",0,"","robustness"],["PhD thesis: SQL Comprehension and Synthesis.","G Obaido","2022","report","scholar.google.co.za","scholar.google.co.za/citations?view_op=view_citation&hl=en&user=cCXONVYAAAAJ&sortby=pubdate&citation_for_view=cCXONVYAAAAJ:Wp0gIr-vW9MC",0,"",""],["PLATO: Predicting Latent Affordances Through Object-Centric Play.","Suneel Belkhale and Dorsa Sadigh","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/belkhale2022plato.pdf",0,"",""],["Playful Interactions for Representation Learning.","Sarah Young and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2107.09046",0,"","policy"],["Politicians must prepare for AI or face the consequences.","Stuart Russell","2022","report","politicshome.com","www.politicshome.com/thehouse/article/politicians-must-prepare-for-ai-or-face-the-consequences",0,"",""],["Pretraining Graph Neural Networks for few-shot Analog Circuit Modeling and Design.","Kourosh Hakhamaneshi and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.15913",0,"","mechanistic-interpretability robustness"],["Probabilistic and Causal Inference: The Works of Judea Pearl.","H and 2 others","2022","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/abstract.html#book7",0,"",""],["Real-World Robot Learning with Masked Visual Pre-training.","Ilija Radosavovic and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.03109",0,"",""],["Reasoning about causal models with infinitely many variables.","J Halpern and 2 others","2022","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/abstract.html#gsem-axiomatization",0,"",""],["Reconstructing the cascade of language processing in the brain using the internal computations of a transformer-based language model.","Kumar and 22 others","2022","report","biorxiv.org","www.biorxiv.org/content/10.1101/2022.06.08.495348v1",0,"",""],["Reducing Exploitability with Population Based Training.","Pavel Czempin and Adam Gleave","2022","report","gleave.me","www.gleave.me/publication/2022-08-reducing-exploitability/",0,"",""],["Reducing Variance in Temporal-Difference Value Estimation via Ensemble of Deep Networks.","Litian Liang and 6 others","2022","paper","ICML 2022","arxiv.org/abs/2209.07670",0,"","benchmarks"],["Reinforcement Learning with Action-Free Pre-Training from Videos.","Younggyo Seo and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.13880",0,"",""],["Relationship to CBT outcome and dropout of decision support tools of the written case formulation, list of treatment goals, and plot of symptom scores.","J Persons and V Gates","2022","report","scholar.google.co.uk","scholar.google.co.uk/citations?view_op=view_citation&hl=en&user=XnUZEcoAAAAJ&sortby=pubdate&citation_for_view=XnUZEcoAAAAJ:LkGwnXOMwfcC",0,"",""],["Reliable Prediction and Decision-Making in Sequential Environments.","P Rashidinejad","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=BgQkdsYAAAAJ&sortby=pubdate&citation_for_view=BgQkdsYAAAAJ:Y0pCki6q_DkC",0,"",""],["Rethinking the purpose of AI.","Stuart Russell","2022","report","ecfr.eu","ecfr.eu/podcasts/episode/rethinking-the-purpose-of-ai-with-stuart-russell/",0,"",""],["Retrospective on the 2021 MineRL BASALT Competition on Learning from Human Feedback.","Rohin Shah and 15 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=odFQXSYAAAAJ&sortby=pubdate&citation_for_view=odFQXSYAAAAJ:8k81kl-MbHgC",0,"","rlhf"],["Reward Uncertainty for Exploration in Preference-based Reinforcement Learning.","Xinran Liang and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.12401",0,"","rlhf benchmarks agents policy"],["Risk Ratios.","Jonathan Stray","2022","report","github.com","github.com/jstray/risk-ratios",0,"",""],["Robotic Weapons Are Coming: What Should We Do About It.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/LAWS.html#:~:text=Robotic%20Weapons%20Are%20Coming%3A%20What%20Should%20We%20Do%20About%20It",0,"",""],["RvS: What is Essential for Offline RL via Supervised Learning?.","Scott Emmons and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2112.10751",0,"","robustness"],["Safety Assurances for Human-Robot Interaction via Confidence-aware Game-theoretic Human Models.","R and 10 others","2022","paper","arXiv preprint","arxiv.org/abs/2109.14700",0,"","evals assurance robustness"],["Scenic4RL: Programmatic Modeling and Generation of Reinforcement Learning Environments.","Abdus Salam Azad and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2106.10365",0,"","benchmarks agents"],["SHARP: Shielding-aware robust planning for safe and efficient human-robot interaction.","H Hu and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:ZeXyd9-uunAC",0,"",""],["Sim-to-Lab-to-Real: Safe RL with Shielding and Generalization Guarantees.","KC Hsu and 4 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=HvjirogAAAAJ&sortby=pubdate&citation_for_view=HvjirogAAAAJ:hFOr9nPyWt4C",0,"",""],["Sim-to-Real 6D Object Pose Estimation via Iterative Self-training for Robotic Bin-picking.","Kai Chen and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.07049",0,"","evals benchmarks"],["Sim-to-Real via Sim-to-Seg: End-to-end Off-road Autonomous Driving Without Real Data.","John So* and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2210.14721",0,"","policy"],["Similarity-based Cooperation.","Caspar Oesterheld and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.14468",0,"","interpretability agents"],["Simplicity as a Cue to Probability: multiple roles for Simplicity in Evaluating Explanations.","TH Vrantsidis and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:FPJr55Dyh1AC",0,"","evals"],["Simplicity beyond probability: Simplicity’s role in evaluating explanations goes beyond providing cues to priors and likelihoods.","T Vrantsidis and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:Ug5p-4gJ2f0C",0,"","evals"],["Social media is polluting society. Moderation alone won’t fix the problem.","Nathaniel Lubinarchive pageThomas Krendl Gilbertarchive page","2022","report","technologyreview.com","www.technologyreview.com/2022/08/09/1057171/social-media-polluting-society-moderation-alone-wont-fix-the-problem/",0,"",""],["Sociotechnical Specification for the Broader Impacts of Autonomous Vehicles.","Thomas Krendl Gilbert and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.07395",0,"","evals"],["Solving Structured Hierarchical Games Using Differential Backward Induction.","Z LI and 6 others","2022","report","strategicreasoning.org","strategicreasoning.org/solving-structured-hierarchical-games-using-differential-backward-induction/",0,"",""],["Spending Thinking Time Wisely: Accelerating MCTS with Virtual Expansions.","Weirui Ye and 2 others","2022","paper","Published at NeurIPS 2022","arxiv.org/abs/2210.12628",0,"","evals"],["Spurious normativity enhances learning of compliance and enforcement behavior in artificial agents.","Raphael Köster and 5 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:4JMBOYKVnBMC",0,"","agents"],["Steerable Partial Differential Operators for Equivariant Neural Networks.","E Jenner and M Weiler","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=de&user=8DgF8HcAAAAJ&sortby=pubdate&citation_for_view=8DgF8HcAAAAJ:d1gkVwhDpl0C",0,"",""],["Stress, Intertemporal Choice, and Mitigation Behavior During the COVID-19 Pandemic.","Agrawal and 10 others","2022","report","doi.org","doi.org/10.31234/osf.io/ureqg",0,"",""],["Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ Tasks..","Yizhong Wang and 39 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.07705",0,"","evals benchmarks"],["SURF: Semi-supervised Reward Learning with Data Augmentation for Feedback-efficient Preference-based Reinforcement Learning.","Jongjin_Park and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.10050",0,"","rlhf agents"],["Teaching Robots to Span the Space of Functional Expressive Motion. .","A and 13 others","2022","report","arjunsripathy.github.io","arjunsripathy.github.io/robot_emotive_space/",0,"",""],["The best of Radio Davos over the last year.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/future/#:~:text=The%20best%20of%20Radio%20Davos%20over%20the%20last%20year",0,"",""],["The Boltzmann Policy Distribution: Accounting for Systematic Suboptimality in Human Models.","Cassidy Laidlaw and Anca Dragan","2022","paper","arXiv preprint","arxiv.org/abs/2204.10759",0,"","policy"],["The Foundations of Artificial Intelligence.","Stuart Russell","2022","report","thegradientpub.substack.com","thegradientpub.substack.com/p/stuart-russell-the-foundations-of#details",0,"",""],["The promises and perils of AI.","Stuart Russell","2022","report","weforum.org","www.weforum.org/agenda/2022/01/artificial-intelligence-stuart-russell-radio-davos",0,"",""],["Time spent thinking in online chess reflects the value of computation.","DMRL\nRussek and 10 others","2022","report","doi.org","doi.org/10.31234/osf.io/8j9zx",0,"",""],["Time-Efficient Reward Learning via Visually Assisted Cluster Ranking.","David Zhang and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2212.00169",0,"","rlhf agents"],["Toward transparent ai: A survey on interpreting the inner structures of deep neural networks.","T Räukur and 3 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:M3NEmzRMIkIC",0,"",""],["Towards more Generalizable One-shot Visual Imitation Learning.","Zhao Mandi and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2110.13423",0,"","evals agents policy"],["Towards Psychologically-Grounded Dynamic Preference Models.","M Curmei and 3 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:blknAaTinKkC",0,"",""],["trading with superintelligence: a wonky proto-alignment scheme","Tamsin Leake","2022","blog","carado.moe","carado.moe/trading-with-superint.html",0,"",""],["Training and Inference on Any-Order Autoregressive Models the Right Way.","Andy Shih and 2 others","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/shih2022aoarm.pdf",0,"",""],["Tuning the Hyperparameters of Anytime Planning: A Metareasoning Approach with Deep RL.","Abhinav Bhatia and 3 others","2022","report","justinsvegliato.com","justinsvegliato.com/pdf/BSNZicaps22.pdf",0,"",""],["Uncertain Decisions Facilitate Better Preference Learning.","Cassidy Laidlaw and Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/research/future/#:~:text=Uncertain%20Decisions%20Facilitate%20Better%20Preference%20Learning",0,"",""],["Uncertainty Estimation for Language Reward Models.","Adam Gleave and Geoffrey Irving","2022","report","gleave.me","www.gleave.me/publication/2022-03-uncertainty-estimation/",0,"",""],["Understanding Value Decomposition Algorithms in Deep Cooperative Multi-Agent Reinforcement Learning.","Z Dou and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=wMjQdBcAAAAJ&sortby=pubdate&citation_for_view=wMjQdBcAAAAJ:IjCSPb-OGe4C",0,"","agents"],["Uni[MASK]: Unified Inference in Sequential Decision Problems. .","M and 22 others","2022","paper","arXiv preprint","arxiv.org/abs/2211.10869",0,"",""],["Using Natural Language and Program Abstractions to Instill Human Inductive Biases in Machines..","DMRL\nKumar and 26 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.11558",0,"","agents"],["Varieties of ignorance: Mystery and the unknown in science and religion.","T Davoodi and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:PoWvk5oyLR8C",0,"",""],["Weakly Supervised Correspondence Learning.","Zihan Wang* and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.00904",0,"","agents"],["What are men and mothers for? The causes and consequences of functional reasoning about social categories.","E Foster-Hanson and T Lombrozo","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:edDO8Oi4QzsC",0,"",""],["When and how children use explanations to guide generalizations.","N Vasil and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=Z4CpYzsAAAAJ&cstart=100&pagesize=100&citation_for_view=Z4CpYzsAAAAJ:-FonjvnnhkoC",0,"",""],["White-Box Adversarial Policies in Deep Reinforcement Learning.","S Casper and 2 others","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=4mVPFQ8AAAAJ&sortby=pubdate&citation_for_view=4mVPFQ8AAAAJ:YFjsv_pBGBYC",0,"",""],["Why we need to regulate non-state use of arms.","Stuart Russell","2022","report","weforum.org","www.weforum.org/agenda/2022/05/regulate-non-state-use-arms",0,"",""],["WordSig: QR streams enabling platform-independent self-identification that’s impossible to deepfake.","A Critch","2022","report","scholar.google.com","scholar.google.com/citations?view_op=view_citation&hl=en&user=F3_yOXUAAAAJ&cstart=20&pagesize=80&citation_for_view=F3_yOXUAAAAJ:QIV2ME_5wuYC",0,"",""],["X-Risk Analysis for AI Research.","Dan Hendrycks and Mantas Mazeika","2022","paper","arXiv preprint","arxiv.org/abs/2206.05862",0,"",""],["Zero-Shot Text-Guided Object Generation with Dream Fields,.","Ajay Jain and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2112.01455",0,"",""],["An extended rocket alignment analogy","remember","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xuYdCDgoBno5haJB6/an-extended-rocket-alignment-analogy",0,"",""],["An Uncanny Prison","Nathan1123","2022","blog","LessWrong","www.lesswrong.com/posts/7PePKKWfzcoqmF3Hs/an-uncanny-prison-1",0,"",""],["Evolution is a bad analogy for AGI: inner alignment","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FyChg3kYG54tEN3u6/evolution-is-a-bad-analogy-for-agi-inner-alignment",0,"",""],["goal-program bricks","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qfopgsFBJLs2u9iww/goal-program-bricks",0,"",""],["Gradient descent doesn't select for inner search","Ivan Vendrov","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TdesHi8kkyokQdDoQ/gradient-descent-doesn-t-select-for-inner-search",0,"",""],["How I think about alignment","Linda Linsefors","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9bpACZn6kG2Ec6CPu/how-i-think-about-alignment",0,"",""],["I missed the crux of the alignment problem the whole time","zeshen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dmcJr2NBg5QuGSHFC/i-missed-the-crux-of-the-alignment-problem-the-whole-time",0,"",""],["Recognition of All Categories of Entities by AI","Hiroshi Yamakawa and Yutaka Matsuo","2022","paper","arXiv preprint","arxiv.org/abs/2208.06590",0,"",""],["Refine's First Blog Post Day","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8jewbj9am99H2oYio/refine-s-first-blog-post-day",0,"",""],["Shapes of Mind and Pluralism in Alignment","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RrirwtP7cNmHtJRxE/shapes-of-mind-and-pluralism-in-alignment",0,"",""],["Steelmining via Analogy","Paul Bricman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/MfCDfuBHXL5ijJFco/steelmining-via-analogy",0,"",""],["The animals and humans analogy for AI risk","freedomandutility","2022","blog","EA Forum","forum.effectivealtruism.org/posts/p6kCzQ2QiWBW6wxCJ/the-animals-and-humans-analogy-for-ai-risk",0,"",""],["The Dumbest Possible Gets There First","Artaxerxes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BPJLzkEpx8Btz9ywq/the-dumbest-possible-gets-there-first",0,"",""],["the Insulated Goal-Program idea","Tamsin Leake","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oTn2PPZLY7a2xJmqh/the-insulated-goal-program-idea",0,"",""],["why my timelines are short: all roads lead to doom","Tamsin Leake","2022","blog","carado.moe","carado.moe/why-timelines-short.html",0,"","forecasting"],["anthropic mindfulness","Tamsin Leake","2022","blog","carado.moe","carado.moe/anthropic-mindfulness.html",0,"",""],["Artificial intelligence wireheading","Big Tony","2022","blog","LessWrong","www.lesswrong.com/posts/HuBRXaqN2FPtQhywr/artificial-intelligence-wireheading",0,"","reward-hacking"],["DeepMind alignment team opinions on AGI ruin arguments","Vika","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qJgz2YapqpFEDTLKn/deepmind-alignment-team-opinions-on-agi-ruin-arguments",0,"",""],["Distillation of The Offense-Defense Balance of Scientific Knowledge","Arjun Yadav","2022","blog","EA Forum","forum.effectivealtruism.org/posts/twmN3eponyRBxsH4P/distillation-of-the-offense-defense-balance-of-scientific",0,"","governance"],["goal-program bricks","Tamsin Leake","2022","blog","carado.moe","carado.moe/goal-program-bricks.html",0,"",""],["Oversight Misses 100% of Thoughts The AI Does Not Think","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/98c5WMDb3iKdzD4tM/oversight-misses-100-of-thoughts-the-ai-does-not-think",0,"",""],["Perfect Predictors","aditya malik","2022","blog","LessWrong","www.lesswrong.com/posts/JXcKqYdcHoabmMpjh/perfect-predictors",0,"","theory"],["Refining the Sharp Left Turn threat model, part 1: claims and mechanisms","Vika and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/usKXS5jGDzjwqv3FJ/refining-the-sharp-left-turn-threat-model-part-1-claims-and",0,"",""],["the foundation book","Tamsin Leake","2022","blog","carado.moe","carado.moe/foundation-book.html",0,"",""],["Timelines explanation post part 1 of ?","Nathan Helm-Burger","2022","blog","LessWrong","www.lesswrong.com/posts/jQXu9tefKfRroegrb/timelines-explanation-post-part-1-of",0,"","forecasting"],["A pseudo mathematical formulation of direct work choice between two x-risks","Joseph Bloom","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FzCNJtat9phGWrWJX/a-pseudo-mathematical-formulation-of-direct-work-choice",0,"",""],["Encultured AI Pre-planning, Part 2: Providing a Service","Andrew_Critch and Nick Hay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2vxoTfuScspraSJeC/encultured-ai-pre-planning-part-2-providing-a-service",0,"",""],["Encultured AI, Part 2: Providing a Service","Andrew Critch and Nick Hay","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MWWZQ8C655iT9zzRd/encultured-ai-part-2-providing-a-service",0,"",""],["future paths","Tamsin Leake","2022","blog","carado.moe","carado.moe/future-paths.html",0,"",""],["Language models seem to be much better than humans at next-token prediction","Buck and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/htrZrxduciZ5QaCjw/language-models-seem-to-be-much-better-than-humans-at-next",0,"",""],["scopes of utopia","Tamsin Leake","2022","blog","carado.moe","carado.moe/utopia-scopes.html",0,"",""],["Seriously, what goes wrong with \"reward the agent when it makes you smile\"?","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/22xf8GmwqGzHbiuLg/seriously-what-goes-wrong-with-reward-the-agent-when-it",0,"","agents"],["Shard Theory: An Overview","David Udell","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xqkGmfikqapbJ2YMj/shard-theory-an-overview",0,"","agents"],["The alignment problem from a deep learning perspective","richard_ngo","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QYbP47ZErrgFYXBLX/the-alignment-problem-from-a-deep-learning-perspective",0,"",""],["what does it mean to value our survival?","Tamsin Leake","2022","blog","carado.moe","carado.moe/value-yourself-surviving.html",0,"",""],["A computational process-tracing method for measuring people’s planning strategies and how they change over time.","Jain and 16 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/jaincomputational.pdf",0,"",""],["active reward learning from multiple teachers.","Peter Barnett1 and 2 others","2022","report","justinsvegliato.com","justinsvegliato.com/s/BFSRsafeai23.pdf",0,"",""],["Active Reward Learning from Multiple Teachers.","Peter Barnett and 3 others","2022","report","rachelfreedman.github.io","rachelfreedman.github.io/assets/Barnett2022.pdf",0,"",""],["An Empirical Investigation of Representation Learning for Imitation.","Cynthia Chen and 11 others","2022","report","openreview.net","openreview.net/forum?id=kBNhgqXatI",0,"",""],["Artificial intelligence development races in heterogeneous settings.","Theodor Cimpeanu and 5 others","2022","report","nature.com","www.nature.com/articles/s41598-022-05729-3",0,"",""],["By how much should Meta's BlenderBot being really bad cause me to update on how justifiable it is for OpenAI and DeepMind to be making significant progress on AI capabilities?","Sisi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/sL8doR3TDjEhNcwGh/by-how-much-should-meta-s-blenderbot-being-really-bad-cause",0,"",""],["Can Humans Do Less-Than-One-Shot Learning?.","Malaviya and 8 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/malaviya2022can.pdf",0,"",""],["Clustering and the efficient use of cognitive resources..","Dasgupta and 4 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/dasgupta2022clustering.pdf",0,"",""],["Cognitive science as a source of forward and inverse models of human decisions for robotics and control.","Ho and 5 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/ho2022cognitive.pdf",0,"",""],["Competence-Aware Systems.","Connor Basich and 5 others","2022","report","justinsvegliato.com","justinsvegliato.com/s/BSWWBZaij22.pdf",0,"",""],["Complex cognitive algorithms preserved by selective social learning in experimental populations.","CEIL\nThompson and 8 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/thompson2022complex.pdf",0,"",""],["Deep models of superficial face judgments.","Peterson and 12 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/petersondeepmodels22.pdf",0,"",""],["Delegation to artificial agents fosters prosocial behaviors in the collective risk dilemma.","Elias Fernández Domingos and 7 others","2022","report","nature.com","www.nature.com/articles/s41598-022-11518-9",0,"","agents"],["Differential Assessment of Black-Box AI Agents.","Rashmeet Kaur Nayyar and 2 others","2022","report","aair-lab.github.io","aair-lab.github.io/Publications/nvs_aaai22.pdf",0,"","agents"],["Distinguishing rule- and exemplar-based generalization in learning systems.","Dasgupta and 6 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/dasgupta22b.pdf",0,"",""],["Dynamic Multi-Robot Task Allocation under Uncertainty and Temporal Constraints.","Shushman Choudhury and 5 others","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/choudhury2022dynamic.pdf",0,"",""],["From partners to populations: A hierarchical Bayesian account of coordination and convention.","Hawkins and 18 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/hawkinspartners.pdf",0,"",""],["Globally inaccurate stereotypes can result from locally adaptive exploration.","Bai and 7 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/bai_globally_2022.pdf",0,"",""],["How Do We Align an AGI Without Getting Socially Engineered? (Hint: Box It)","Peter S. Park and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/p62bkNAciLsv6WFnR/how-do-we-align-an-agi-without-getting-socially-engineered",0,"",""],["How Do We Align an AGI Without Getting Socially Engineered? (Hint: Box It)","Peter S. Park and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/p62bkNAciLsv6WFnR/how-do-we-align-an-agi-without-getting-socially-engineered",0,"",""],["How much alignment data will we need in the long run?","Jacob_Hilton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qoz2ryN4GDqEWPBnQ/how-much-alignment-data-will-we-need-in-the-long-run-1",0,"",""],["How To Go From Interpretability To Alignment: Just Retarget The Search","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/w4aeAFzSAguvqA5qu/how-to-go-from-interpretability-to-alignment-just-retarget",0,"","interpretability"],["If it’s important, then I’m curious: Increasing perceived usefulness stimulates curiosity.","Dubey and 6 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/dubeyimportantcurious.pdf",0,"",""],["Inferring strategies from observations in long iterated Prisoner’s dilemma experiments.","Eladio Montero-Porras and 2 others","2022","report","nature.com","www.nature.com/articles/s41598-022-11654-2",0,"",""],["Is the rise of killer machines closer than we think?.","Stuart Russell","2022","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/times22-russell-intvw.pdf",0,"",""],["Leveraging artificial intelligence to improve people’s planning strategies. Proceedings of the National Academy of Sciences.","Callaway and 22 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/callawayleveraging.pdf",0,"",""],["Memory transmission in small groups and large networks: An empirical study.","Gates and 7 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/gates2022memory.pdf",0,"",""],["Multiscale Heterogeneous Optimal Lockdown Control for COVID-19 Using Geographic Information.","C and 15 others","2022","report","nature.com","www.nature.com/articles/s41598-022-07692-5",0,"",""],["Natural Selection Favors AIs over Humans.","Dan Hendrycks","2022","report","drive.google.com","drive.google.com/file/d/1p4ZAuEYHL_21tqstJOGsMiG4xaRBtVcj/view?usp=share_link",0,"",""],["OpenOOD: Benchmarking Generalized Out-of-Distribution Detection.","Jingkang Yang and 15 others","2022","report","openreview.net","openreview.net/pdf?id=gT6j4_tskUt",0,"","benchmarks"],["Optimal policies for free recall.","Zhang and 7 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/zhangoptimal.pdf",0,"",""],["Overcoming Individual Limitations Through Distributed Computation: Rational Information Accumulation in Multigenerational Populations..","Hardy and 10 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/hardy2022overcoming.pdf",0,"",""],["People construct simplified mental representations to plan..","DMRL\nHo and 16 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/ho2022people.pdf",0,"",""],["Possible directions in AI ideal governance research","RoryG","2022","blog","EA Forum","forum.effectivealtruism.org/posts/F2DkHdKS8G3tD4CaG/possible-directions-in-ai-ideal-governance-research",0,"","governance"],["Predicting Human Similarity Judgments Using Large Language Models..","Marjieh and 11 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/marjieh2022predicting.pdf",0,"",""],["Probing BERT’s priors with serial reproduction chains.","SML\nYamakoshi and 7 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/yamakoshiprobing.pdf",0,"","interpretability"],["Rational heuristics for one-shot games.","Callaway and 6 others","2022","report","gustavkarreskog.com","gustavkarreskog.com/files/jmp_karreskog.pdf",0,"",""],["Rational use of cognitive resources in human planning. Nature Human Behaviour,.","Callaway and 15 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/callawayrationaluse.pdf",0,"",""],["Selecting the Partial State Abstractions of MDPs: A Metareasoning Approach with Deep Reinforcement Learning.","Samer B and 5 others","2022","report","justinsvegliato.com","justinsvegliato.com/s/NSBRZiros2022.pdf",0,"",""],["Selecting the Partial State Abstractions of MDPs: A Metareasoning Approach with Deep Reinforcement Learning.","Samer B and 5 others","2022","report","static1.squarespace.com","static1.squarespace.com/static/6266e3c48fb0751e74f60eb6/t/62e6d6afb0c92718ae985800/1659295408089/NSBRZiros2022.pdf",0,"",""],["Shades of confusion: Lexical uncertainty modulates ad hoc coordination in an interactive communication task..","SML\nMurthy and 8 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/murthyshades.pdf",0,"",""],["Shared Autonomy for Robotic Manipulation with Language Corrections.","Siddharth Karamcheti* and 4 others","2022","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/karamcheti2022lilac.pdf",0,"",""],["The alignment problem from a deep learning perspective","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/KbyRPCAsWv5GtfrbG/the-alignment-problem-from-a-deep-learning-perspective",0,"",""],["The experimental evolution of human culture: flexibility, fidelity and environmental instability.","Morgan and 8 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/morgan2022experimental.pdf",0,"",""],["The History, Epistemology and Strategy of Technological Restraint, and lessons for AI (short essay)","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pJuS5iGbazDDzXwJN/the-history-epistemology-and-strategy-of-technological",0,"","governance"],["the Insulated Goal-Program idea","Tamsin Leake","2022","blog","carado.moe","carado.moe/insulated-goal-program.html",0,"",""],["The pursuit of happiness: A reinforcement learning perspective on habituation and comparisons.","Dubey and 6 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/dubey2022pursuit.pdf",0,"",""],["There are two factions working to prevent AI dangers. Here’s why they’re deeply divided.","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AARnvz99hiEytnA9k/there-are-two-factions-working-to-prevent-ai-dangers-here-s",0,"",""],["Trade Regulation Rule on Commercial Surveillance and Data Security Rulemaking.","Thomas Krendl Gilbert and Micah Carroll","2022","report","thomaskrendlgilbert.com","www.thomaskrendlgilbert.com/uploads/1/2/1/2/121285828/ftc_final.pdf",0,"","governance"],["Tuning the Hyperparameters of Anytime Planning: A Metareasoning Approach with Deep Reinforcement Learning.","Abhinav Bhatia and 4 others","2022","report","static1.squarespace.com","static1.squarespace.com/static/6266e3c48fb0751e74f60eb6/t/626ed6898ec65809d49641cb/1651431052820/BSNZicaps22.pdf",0,"",""],["Understanding Recommenders..","J Stray","2022","report","medium.com","medium.com/understanding-recommenders",0,"",""],["unviable moral patients","Tamsin Leake","2022","blog","carado.moe","carado.moe/unviable-moral-patient.html",0,"",""],["Using GPT-3 to augment human intelligence","Henrik Karlsson","2022","blog","LessWrong","www.lesswrong.com/posts/spBoxzcaCrqXqyQHq/using-gpt-3-to-augment-human-intelligence",0,"","scalable-oversight"],["Using Natural Language to Guide Meta-Learning Agents towards Human-like Inductive Biases.","Kumar and 21 others","2022","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/kumarusing.pdf",0,"","agents"],["Voluntary safety commitments provide an escape from over-regulation in AI development.","The Anh Han and 4 others","2022","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0160791X21003183?dgcid=author",0,"","governance"],["Announcing: Mechanism Design for AI Safety - Reading Group","Rubi J. Hudson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FhqZZFydyQG9WTSKR/announcing-mechanism-design-for-ai-safety-reading-group",0,"",""],["Are ya winning, son?","Nathan1123","2022","blog","LessWrong","www.lesswrong.com/posts/2KvjY6HZ64QXDE2T6/are-ya-winning-son",0,"","theory"],["Bahamian Adventures: An Epic Tale of Entrepreneurship, AI Strategy Research and Potatoes","Jaime Sevilla","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Ekso4kAkjLivnaaQP/bahamian-adventures-an-epic-tale-of-entrepreneurship-ai",0,"",""],["Content generation. Where do we draw the line?","Q Home","2022","blog","LessWrong","www.lesswrong.com/posts/TaqBzqhzEPi8eHtC2/content-generation-where-do-we-draw-the-line",0,"",""],["Effective Persuasion For AI Alignment Risk","Brian Lui","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9rdkqNd2faqzP9f9p/effective-persuasion-for-ai-alignment-risk",0,"",""],["How would two superintelligent AIs interact, if they are unaligned with each other?","Nathan1123","2022","blog","LessWrong","www.lesswrong.com/posts/QEMbewiGaypjfmDi7/how-would-two-superintelligent-ais-interact-if-they-are",0,"","theory"],["How/When Should One Introduce AI Risk Arguments to People Unfamiliar With the Idea?","Harrison Durland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Hw7DjsX6xjCAxXgGv/how-when-should-one-introduce-ai-risk-arguments-to-people",0,"",""],["ruling out intuitions about materially acausal things","Tamsin Leake","2022","blog","carado.moe","carado.moe/ruling-out-intuitions-materially-acausal-intuitions.html",0,"",""],["Spicy takes about AI policy (Clark, 2022)","Will Aldred","2022","blog","EA Forum","forum.effectivealtruism.org/posts/mSdnDYzfqh5MEYgox/spicy-takes-about-ai-policy-clark-2022",0,"","governance policy"],["Which of these arguments for x-risk do you think we should test?","Wim","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hFLEpodjWZvQLgMza/which-of-these-arguments-for-x-risk-do-you-think-we-should",0,"",""],["\"Normal accidents\" and AI systems","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yxk2ue2eLeCrozvRz/normal-accidents-and-ai-systems",0,"","governance"],["Classifying sources of AI x-risk","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/e55QpEExmtkRjw9CD/classifying-sources-of-ai-x-risk",0,"",""],["Disagreements about Alignment: Why, and how, we should try to solve them","ojorgensen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xfc6x9FbiK2kEorRo/disagreements-about-alignment-why-and-how-we-should-try-to",0,"",""],["Encultured AI Pre-planning, Part 1: Enabling New Benchmarks","Andrew_Critch and Nick Hay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AR6mfydDJiGksj6Co/encultured-ai-pre-planning-part-1-enabling-new-benchmarks",0,"","benchmarks"],["Encultured AI, Part 1 Appendix: Relevant Research Examples","Andrew_Critch and Nick Hay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PvuuBN39pmjw6wRpj/encultured-ai-part-1-appendix-relevant-research-examples",0,"",""],["Encultured AI, Part 1: Enabling New Benchmarks","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yczkGfcfWoRN6zfrf/encultured-ai-part-1-enabling-new-benchmarks",0,"","benchmarks"],["Future Matters #4: AI timelines, AGI risk, and existential risk from climate change","Pablo and matthew.vandermerwe","2022","blog","EA Forum","forum.effectivealtruism.org/posts/XQbhnKgXiRTv4vfxt/future-matters-4-ai-timelines-agi-risk-and-existential-risk",0,"","forecasting"],["General alignment properties","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FMdGt9S9irgxeD9Xz/general-alignment-properties",0,"","theory"],["How technical safety standards could promote TAI safety","Cullen and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zvbGXCxc5jBowCuNX/how-technical-safety-standards-could-promote-tai-safety",0,"","evals governance policy"],["Interpretability/Tool-ness/Alignment/Corrigibility are not Composable","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qXtbBAxmFkAQLQEJE/interpretability-tool-ness-alignment-corrigibility-are-not",0,"","interpretability"],["Steganography in Chain of Thought Reasoning","A Ray","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yDcMDJeSck7SuBs24/steganography-in-chain-of-thought-reasoning",0,"","chain-of-thought-faithfulness"],["Will Superhuman AI be created?","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/argument-for-likelihood-of-superhuman-ai/",0,"",""],["How I Came To Longtermism On My Own & An Outsider Perspective On EA Longtermism","Jordan Arel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/bwn3zPFkfesNhizCa/how-i-came-to-longtermism-on-my-own-and-an-outsider",0,"",""],["How would Logical Decision Theories address the Psychopath Button?","Nathan1123","2022","blog","LessWrong","www.lesswrong.com/posts/JpAXF8R6pAXhFfZuj/how-would-logical-decision-theories-address-the-psychopath",0,"","theory"],["Jack Clark on the realities of AI policy","Kaj_Sotala","2022","blog","LessWrong","www.lesswrong.com/posts/R3tXGhSCgYbp3kXm2/jack-clark-on-the-realities-of-ai-policy",0,"","governance policy"],["List of sources arguing against existential risk from AI","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/list-of-sources-arguing-against-existential-risk-from-ai/",0,"",""],["Longtermists Should Work on AI - There is No \"AI Neutral\" Scenario","simeon_c and Amber Dawn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/q2zNoDbphDgscTdAF/longtermists-should-work-on-ai-there-is-no-ai-neutral",0,"",""],["Why does no one care about AI?","Olivia Addy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Gah8junjra4cTN9G8/why-does-no-one-care-about-ai",0,"",""],["A Data limited future","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gqqhYijxcKAtuAFjL/a-data-limited-future",0,"",""],["AI risks: the most convincing argument","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/aQ6QP3rLsLcZqYodr/ai-risks-the-most-convincing-argument",0,"",""],["Announcing the Introduction to ML Safety course","Dan H and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/4F8Bg8Z5cePTBofzo/announcing-the-introduction-to-ml-safety-course",0,"",""],["Announcing the Introduction to ML Safety Course","ThomasW and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/GQqbbEJzBd4GsraPT/announcing-the-introduction-to-ml-safety-course",0,"",""],["Collection of work on 'Should you focus on the EU if you're interested in AI governance for longtermist/x-risk reasons?'","MichaelA","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yNxn4HxDSMdRyrv6E/collection-of-work-on-should-you-focus-on-the-eu-if-you-re",0,"","governance policy"],["Incentives to create AI systems known to pose extinction risks","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/incentives-to-create-x-risky-ai-systems/",0,"",""],["List of sources arguing for existential risk from AI","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/list-of-sources-arguing-for-existential-risk-from-ai/",0,"",""],["probability under potential hardware failure","Tamsin Leake","2022","blog","carado.moe","carado.moe/probability-hardware-failure.html",0,"",""],["quantum immortality and local deaths under X-risk","Tamsin Leake","2022","blog","carado.moe","carado.moe/quantum-immortality-local-deaths.html",0,"",""],["Why I Am Skeptical of AI Regulation as an X-Risk Mitigation Strategy","A Ray","2022","blog","LessWrong","www.lesswrong.com/posts/zQnzhGLDp2PSvAjcW/why-i-am-skeptical-of-ai-regulation-as-an-x-risk-mitigation",0,"","governance"],["$20K In Bounties for AI Safety Public Materials","Dan H and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gWM8cgZgZ9GQAYTqF/usd20k-in-bounties-for-ai-safety-public-materials",0,"",""],["$20K in Bounties for AI Safety Public Materials","ThomasW and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JsS5vuiHEoBMbYk5R/usd20k-in-bounties-for-ai-safety-public-materials",0,"",""],["Bridging Expected Utility Maximization and Optimization","Whispermute","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rQDYQrDjPGqjrf8Mk/bridging-expected-utility-maximization-and-optimization",0,"","agents theory"],["Counterfactuals are Confusing because of an Ontological Shift","Chris_Leong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/MiWBC2A2cADspausG/counterfactuals-are-confusing-because-of-an-ontological",0,"",""],["Rant on Problem Factorization for Alignment","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tmuFmHuyb4eWmPXz8/rant-on-problem-factorization-for-alignment",0,"",""],["Where are the red lines for AI?","Karl von Wendt","2022","blog","LessWrong","www.lesswrong.com/posts/eStLg3uhHmzjCqWDm/where-are-the-red-lines-for-ai",0,"","governance"],["2022 AI expert survey results","Zach Stein-Perlman","2022","blog","EA Forum","forum.effectivealtruism.org/posts/mjB9osLTJJM4zKhoq/2022-ai-expert-survey-results",0,"","forecasting"],["2022 Expert Survey on Progress in AI","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/2022-expert-survey-on-progress-in-ai/",0,"",""],["Announcing the SPT Model Web App for AI Governance","Paolo Bova and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/c73nsggC2GQE5wBjq/announcing-the-spt-model-web-app-for-ai-governance",0,"","governance forecasting"],["Convergence Towards World-Models: A Gears-Level Model","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HzSdYWvdrdQqG9tqW/convergence-towards-world-models-a-gears-level-model",0,"",""],["Does China have AI alignment resources/institutions? How can we prioritize creating more?","JakubK","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eQa4WtedcAookJ7nM/does-china-have-ai-alignment-resources-institutions-how-can",0,"","governance"],["Surprised by ELK report's counterexample to Debate, IDA","Evan R. Murphy","2022","blog","LessWrong","www.lesswrong.com/posts/Qsc3G2HemFWLobDSw/surprised-by-elk-report-s-counterexample-to-debate-ida",0,"","scalable-oversight eliciting-latent-knowledge"],["The Pragmascope Idea","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gdEDPHjCY5DKsMsvE/the-pragmascope-idea",0,"",""],["Values and control","dotsam","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eqnDKGjaujNWN3t3i/values-and-control",0,"",""],["What do ML researchers think about AI in 2022?","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/what-do-ml-researchers-think-about-ai-in-2022/",0,"",""],["What do ML researchers think about AI in 2022?","KatjaGrace","2022","blog","LessWrong","www.lesswrong.com/posts/H6hMugfY3tDQGfqYL/what-do-ml-researchers-think-about-ai-in-2022",0,"","forecasting"],["Why we need a new agency to regulate advanced artificial intelligence","Michael Huang","2022","blog","EA Forum","forum.effectivealtruism.org/posts/GfdDZBiFjBb5fogCN/why-we-need-a-new-agency-to-regulate-advanced-artificial",0,"","governance policy"],["Would \"Manhattan Project\" style be beneficial or deleterious for AI Alignment?","Just Learning","2022","blog","LessWrong","www.lesswrong.com/posts/bwcEQ4zRmLbbhAhA9/would-manhattan-project-style-be-beneficial-or-deleterious",0,"","governance"],["Ajeya's TAI timeline shortened from 2050 to 2040","Zach Stein-Perlman","2022","blog","EA Forum","forum.effectivealtruism.org/posts/M6NwNYBMkgn7eyZZR/ajeya-s-tai-timeline-shortened-from-2050-to-2040",0,"","forecasting"],["AMA: Ought","stuhlmueller and jungofthewon","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YBaJvhcat3PGhCCnk/ama-ought",0,"",""],["Externalized reasoning oversight: a research direction for language model alignment","tamera","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FRRb6Gqem8k69ocbi/externalized-reasoning-oversight-a-research-direction-for",0,"","chain-of-thought-faithfulness"],["Precursor checking for deceptive alignment","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pRt4E3nmPBtWZiT4A/precursor-checking-for-deceptive-alignment",0,"","interpretability alignment-faking deception"],["Three pillars for avoiding AGI catastrophe: Technical alignment, deployment decisions, and coordination","alexlintz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eggdG27y75ot8dNn7/three-pillars-for-avoiding-agi-catastrophe-technical",0,"","governance"],["Three pillars for avoiding AGI catastrophe: Technical alignment, deployment decisions, and coordination","Alex Lintz","2022","blog","LessWrong","www.lesswrong.com/posts/cm5dCKYCamotzEMxq/three-pillars-for-avoiding-agi-catastrophe-technical",0,"","governance"],["tiling the cosmos might be unavoidable","Tamsin Leake","2022","blog","carado.moe","carado.moe/tiling-unavoidable.html",0,"",""],["What if AI development goes well?","RoryG","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9EjMoD8BRhXEsfzMh/what-if-ai-development-goes-well-3",0,"","governance"],["Exploratory survey on psychology of AI risk perception","Daniel_Friedrich","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ygnYXvkezCLasdh7A/exploratory-survey-on-psychology-of-ai-risk-perception",0,"",""],["Information in risky technology races","nemeryxu","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LbZN3YzXHe357EjcJ/information-in-risky-technology-races",0,"","governance policy"],["isn't it weird that we have a chance at all?","Tamsin Leake","2022","blog","carado.moe","carado.moe/weird-chance.html",0,"",""],["Law-Following AI 4: Don't Rely on Vicarious Liability","Cullen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HDmcJv6SdyEFpFbcD/law-following-ai-4-don-t-rely-on-vicarious-liability",0,"","governance"],["Two-year update on my personal AI timelines","Ajeya Cotra","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AfH2oPHCApdKicM4m/two-year-update-on-my-personal-ai-timelines",0,"","forecasting"],["Announcing the GovAI Policy Team","MarkusAnderljung","2022","blog","EA Forum","forum.effectivealtruism.org/posts/jatnoouJuCpcKpnoh/announcing-the-govai-policy-team",0,"","governance policy"],["Few-shot Adaptation Works with UnpredicTable Data","Jun Shern Chan","2022","paper","arXiv preprint","arxiv.org/abs/2208.01009",0,"",""],["Preventing an AI-related catastrophe","Benjamin Hilton","2022","report","80000hours.org","80000hours.org/problem-profiles/artificial-intelligence/",0,"",""],["chinchilla's wild implications","nostalgebraist","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Fpvch8RR29qLEWNH/chinchilla-s-wild-implications",0,"","scaling-laws"],["Wanted: Notation for credal resilience","PeterH","2022","blog","LessWrong","www.lesswrong.com/posts/uGqdJCrqznzLBDXcr/wanted-notation-for-credal-resilience",0,"","forecasting theory"],["AI timelines by bio anchors: the debate in one place","Will Aldred","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NnygBgntvoGSuvsRH/ai-timelines-by-bio-anchors-the-debate-in-one-place",0,"","forecasting"],["How transparency changed over time","ViktoriaMalyasova","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ngwNHAy5TjStZnJzQ/how-transparency-changed-over-time",0,"","interpretability"],["July 2022 Newsletter","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/07/30/july-2022-newsletter/",0,"",""],["Abstracting The Hardness of Alignment: Unbounded Atomic Optimization","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/uhxpJyGYQ5FQRvdjY/abstracting-the-hardness-of-alignment-unbounded-atomic",0,"",""],["Closing the Feedback Loop on AI Safety Research.","Ben.Hartley","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3tkYQi7eyHnARzfPu/closing-the-feedback-loop-on-ai-safety-research",0,"",""],["Comparing Four Approaches to Inner Alignment","Lucas Teixeira","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/KWmrz9WbGntMGMb73/comparing-four-approaches-to-inner-alignment",0,"",""],["Conjecture: Internal Infohazard Policy","Connor Leahy and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gs29k3beHiqWFZqnn/conjecture-internal-infohazard-policy",0,"","policy"],["Humans Reflecting on HRH","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XsYAL4jvzztomSjKy/humans-reflecting-on-hrh",0,"",""],["AI Alignment is intractable (and we humans should stop working on it)","GPT 3","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dPh5FgqwuQGGA6FSr/ai-alignment-is-intractable-and-we-humans-should-stop",0,"",""],["Announcing the AI Safety Field Building Hub, a new effort to provide AISFB projects, mentorship, and funding","Vael Gates","2022","blog","LessWrong","www.lesswrong.com/posts/tD4bNRzHXa2Th7yPs/announcing-the-ai-safety-field-building-hub-a-new-effort-to",0,"",""],["Efficient training of language models to fill in the middle","OpenAI Research","2022","blog","openai.com","openai.com/research/efficient-training-of-language-models-to-fill-in-the-middle",0,"",""],["Latent Properties of Lifelong Learning Systems","Corban Rivera and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.14378",0,"",""],["Safety without oppression: an AI governance problem","Nathan_Barnard","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LTCwe2RaCreLZ4gd2/safety-without-oppression-an-ai-governance-problem",0,"","governance"],["Toward Supporting Perceptual Complementarity in Human-AI Collaboration via Reflection on Unobservables","Kenneth Holstein and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.13834",0,"",""],["AGI ruin scenarios are likely (and disjunctive)","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ervaGwJ2ZcwqfCcLx/agi-ruin-scenarios-are-likely-and-disjunctive",0,"",""],["FLI is hiring a new Director of US Policy","aaguirre","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9orJx6uvgbLD7FkGC/fli-is-hiring-a-new-director-of-us-policy",0,"","governance policy"],["How long does it take to undersrand AI X-Risk from scratch so that I have a confident, clear mental model of it from first principles?","Jordan Arel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ajzK5mNTDdxMzx36d/how-long-does-it-take-to-undersrand-ai-x-risk-from-scratch",0,"",""],["Levels of Pluralism","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wi3upQibefMcFs5to/levels-of-pluralism",0,"",""],["Moral strategies at different capability levels","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jDQm7YJxLnMnSNHFu/moral-strategies-at-different-capability-levels",0,"","theory"],["Principles of Privacy for Alignment Research","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SsuqYoyBnheSj7jLw/principles-of-privacy-for-alignment-research",0,"",""],["Toward Transparent AI: A Survey on Interpreting the Inner Structures of Deep Neural Networks","Tilman Räuker and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.13243",0,"","interpretability benchmarks robustness"],["Unifying Bargaining Notions (2/2)","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RZNmNwc9SxdKayeQh/unifying-bargaining-notions-2-2",0,"",""],["Active Inference as a formalisation of instrumental convergence","Roman Leventov","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ostLZyhnBPndno2zP/active-inference-as-a-formalisation-of-instrumental",0,"","instrumental-convergence"],["an anthropics example","Tamsin Leake","2022","blog","carado.moe","carado.moe/anthropics-example.html",0,"",""],["How much should you optimize for the short-timelines scenario?","SoerenMind","2022","blog","EA Forum","forum.effectivealtruism.org/posts/evakA8beTDq4KbxFy/how-much-should-you-optimize-for-the-short-timelines",0,"","forecasting"],["Humanity’s vast future and its implications for cause prioritization","BrownHairedEevee","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DZ8JFxWo4tzuj6L85/humanity-s-vast-future-and-its-implications-for-cause",0,"",""],["Neartermists should consider AGI timelines in their spending decisions","Tristan Cook","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ebYdBNpGnshhm2Gkq/neartermists-should-consider-agi-timelines-in-their-spending",0,"","forecasting"],["NeurIPS ML Safety Workshop 2022","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pY4J2qNaHgKp2nbEd/neurips-ml-safety-workshop-2022",0,"",""],["Slowing down AI progress?","Eleni_A","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YW6fDEDsd3MXDKhYD/slowing-down-ai-progress",0,"","governance"],["«Boundaries», Part 1: a key missing concept from utility theory","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8oMF8Lv5jiGaQSFvo/boundaries-part-1-a-key-missing-concept-from-utility-theory",0,"",""],["A hazard analysis framework for code synthesis large language models","OpenAI Research","2022","blog","openai.com","openai.com/research/a-hazard-analysis-framework-for-code-synthesis-large-language-models",0,"",""],["AGI Safety Needs People With All Skillsets!","Severin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NJtC8xzD8BgF3TmEp/agi-safety-needs-people-with-all-skillsets",0,"",""],["Does agent foundations cover all future ML systems?","Jonas Hallgren","2022","blog","LessWrong","www.lesswrong.com/posts/Fk3KYMxGLzDnwjFzo/does-agent-foundations-cover-all-future-ml-systems",0,"","agents theory"],["How much should we worry about mesa-optimization challenges?","sudo -i","2022","blog","LessWrong","www.lesswrong.com/posts/apFCckw6grxBH7bYL/how-much-should-we-worry-about-mesa-optimization-challenges",0,"",""],["Reward is not the optimization target","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pdaGN6pQyQarFHXF4/reward-is-not-the-optimization-target",0,"","reward-hacking"],["Unifying Bargaining Notions (1/2)","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rYDas2DDGGDRc8gGB/unifying-bargaining-notions-1-2",0,"","theory"],["Brainstorm of things that could force an AI team to burn their lead","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/p3s8RvkcyTwzu27ps/brainstorm-of-things-that-could-force-an-ai-team-to-burn",0,"",""],["Finding Skeletons on Rashomon Ridge","David Udell and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/PJhvcTkwpGr9Ysmcd/finding-skeletons-on-rashomon-ridge",0,"","interpretability"],["We Did AGISF’s 8-week Course in 3 Days. Here’s How it Went","ag4000 and Logan Riggs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cgcSoDqGYsiBfPEy4/we-did-agisf-s-8-week-course-in-3-days-here-s-how-it-went",0,"",""],["Robustness to Scaling Down: More Important Than I Thought","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pA3F9oejzvGg6Kf3a/robustness-to-scaling-down-more-important-than-i-thought",0,"","robustness"],["Which singularity schools plus the no singularity school was right?","Noosphere89","2022","blog","LessWrong","www.lesswrong.com/posts/yenr6Zp83PHd6Beab/which-singularity-schools-plus-the-no-singularity-school-was",0,"","forecasting"],["Connor Leahy on Conjecture and Dying with Dignity","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QR7yGoFBonY6hege9/connor-leahy-on-conjecture-and-dying-with-dignity",0,"",""],["Maybe AI risk shouldn't affect your life plan all that much","Justis","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wQAYidiuiC42h4BKX/maybe-ai-risk-shouldn-t-affect-your-life-plan-all-that-much",0,"",""],["Reasons I’ve been hesitant about high levels of near-ish AI risk","elifland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5hprBzprm7JjJTHNX/reasons-i-ve-been-hesitant-about-high-levels-of-near-ish-ai-1",0,"",""],["[AN #173] Recent language model results from DeepMind","Rohin Shah","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HXDkCtk9tae5wFmjG/an-173-recent-language-model-results-from-deepmind",0,"",""],["A \"Solipsistic\" Repugnant Conclusion","Ramiro","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TLbiusoP77D9vCnzJ/a-solipsistic-repugnant-conclusion",0,"",""],["Conditioning Generative Models with Restrictions","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/adiszfnFgPEnRsGSr/conditioning-generative-models-with-restrictions",0,"",""],["How much to optimize for the short-timelines scenario?","SoerenMind","2022","blog","LessWrong","www.lesswrong.com/posts/iJYzREEsuG8g95pvC/how-much-to-optimize-for-the-short-timelines-scenario",0,"","forecasting"],["Our Existing Solutions to AGI Alignment (semi-safe)","Michael Soareverix","2022","blog","LessWrong","www.lesswrong.com/posts/ubQDcDxjNJ2Exp3ni/our-existing-solutions-to-agi-alignment-semi-safe",0,"",""],["UK AI Policy Report: Content, Summary, and its Impact on EA Cause Areas","Algo_Law","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EyJEL84MGz9KAAyx9/uk-ai-policy-report-content-summary-and-its-impact-on-ea",0,"","governance policy"],["Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations","Haoran Xu and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.10050",0,"","agents policy"],["How to Diversify Conceptual AI Alignment: the Model Behind Refine","adamShimi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/7zXEEQED89Ebbazc2/how-to-diversify-conceptual-ai-alignment-the-model-behind",0,"",""],["How to Diversify Conceptual Alignment: the Model Behind Refine","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5uiQkyKdejX3aEHLM/how-to-diversify-conceptual-alignment-the-model-behind",0,"",""],["Social scientists interested in AI safety should consider doing direct technical AI safety research, (possibly meta-research), or governance, support roles, or community building instead","Vael Gates","2022","blog","EA Forum","forum.effectivealtruism.org/posts/WHDb9r9yMFetG7oz5/social-scientists-interested-in-ai-safety-should-consider",0,"","governance"],["The Need for a Meta-Architecture for Robot Autonomy","Stalin Muñoz Gutiérrez and Gerald Steinbauer-Wagner","2022","paper","EPTCS 362, 2022, pp. 81-97","arxiv.org/abs/2207.09712",0,"","agents"],["A Critique of AI Alignment Pessimism","ExCeph","2022","blog","LessWrong","www.lesswrong.com/posts/vrcrsd5svM3riDFst/a-critique-of-ai-alignment-pessimism",0,"","instrumental-convergence governance"],["Abram Demski's ELK thoughts and proposal - distillation","Rubi J. Hudson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kF74mHH6SujRoEEFA/abram-demski-s-elk-thoughts-and-proposal-distillation",0,"","eliciting-latent-knowledge"],["Bounded complexity of solving ELK and its implications","Rubi J. Hudson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8cr9fJnay97GEYPt3/bounded-complexity-of-solving-elk-and-its-implications",0,"","eliciting-latent-knowledge"],["Help ARC evaluate capabilities of current language models (still need people)","Beth Barnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/e3j7h4mPHvkRynbco/help-arc-evaluate-capabilities-of-current-language-models",0,"","evals"],["What I'm doing","Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/jcwm3bazs2sj686KC/what-i-m-doing",0,"",""],["A distillation of Evan Hubinger's training stories (for SERI MATS)","Daphne_W","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wPLeBqsLgJyFyuTr7/a-distillation-of-evan-hubinger-s-training-stories-for-seri",0,"",""],["A Survey of the Potential Long-term Impacts of AI","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3ffgjMEJ4jY4rdgJy/a-survey-of-the-potential-long-term-impacts-of-ai",0,"",""],["Boolean Decision Rules for Reinforcement Learning Policy Summarisation","James McCarthy and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.08651",0,"","evals agents policy"],["Conditioning Generative Models for Alignment","Jozdien","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JqnkeqaPseTgxLgEL/conditioning-generative-models-for-alignment",0,"","automated-alignment-research"],["Deception?! I ain’t got time for that!","Paul Colognese","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/C8XTFtiA5xtje6957/deception-i-ain-t-got-time-for-that",0,"","deception"],["Forecasting ML Benchmarks in 2023","jsteinhardt","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/arveXgFbJwascKtQC/forecasting-ml-benchmarks-in-2023",0,"","benchmarks forecasting"],["GPT-2 as step toward general intelligence (Alexander, 2019)","Will Aldred","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gw3tyZShzig28B4PE/gpt-2-as-step-toward-general-intelligence-alexander-2019",0,"",""],["How Interpretability can be Impactful","Connall Garrod","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cj4hWE2xBf7t8nKkk/how-interpretability-can-be-impactful",0,"","interpretability"],["Quantilizers and Generative Models","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tz3hoCs2efHjzNYm5/quantilizers-and-generative-models",0,"",""],["Training goals for large language models","Johannes Treutlein","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dWJNFHnC4bkdbovug/training-goals-for-large-language-models",0,"","theory"],["What’s so dangerous about AI anyway? – Or: What it means to be a superintelligence","Thomas Kehrenberg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/rYNGXyCBFQSGupqJA/what-s-so-dangerous-about-ai-anyway-or-what-it-means-to-be-a",0,"",""],["Why EAs are skeptical about AI Safety","Lukas Trötzmüller","2022","blog","EA Forum","forum.effectivealtruism.org/posts/8JazqnCNrkJtK2Bx4/why-eas-are-skeptical-about-ai-safety",0,"",""],["Without specific countermeasures, the easiest path to transformative AI likely leads to AI takeover","Ajeya Cotra","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pRkFkzwKZ2zfa3R6H/without-specific-countermeasures-the-easiest-path-to",0,"","situational-awareness"],["Without specific countermeasures, the easiest path to transformative AI likely leads to AI takeover","Ajeya","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Y3sWcbcF7np35nzgu/without-specific-countermeasures-the-easiest-path-to-1",0,"",""],["Do EA folks think that a path to zero AGI development is feasible or worthwhile for safety from AI?","Noah Scales","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tybiN9M8rnFrDF8ed/do-ea-folks-think-that-a-path-to-zero-agi-development-is",0,"",""],["Examples of AI Increasing AI Progress","ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/W3tZacTRt4koHyxbr/examples-of-ai-increasing-ai-progress",0,"",""],["Explore Risks from Emerging Technology with Peers Outside of (or New to) the AI Alignment Community - Express Interest by August 8","Fasori","2022","blog","EA Forum","forum.effectivealtruism.org/posts/rpDvh72yvyN8yfPQL/explore-risks-from-emerging-technology-with-peers-outside-of",0,"",""],["Four questions I ask AI safety researchers","Akash","2022","blog","EA Forum","forum.effectivealtruism.org/posts/c2yZNSwvccJGrjmMM/four-questions-i-ask-ai-safety-researchers",0,"",""],["Why I Think Abrupt AI Takeoff","lincolnquirk","2022","blog","LessWrong","www.lesswrong.com/posts/wv8cKEyaRqLfHqHzZ/why-i-think-abrupt-ai-takeoff",0,"","forecasting"],["Why you might expect homogeneous take-off: evidence from ML research","Andrei Alexandru","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/RQn45KzN5cojLLb3L/why-you-might-expect-homogeneous-take-off-evidence-from-ml",0,"","forecasting"],["Alignment as Game Design","Shoshannah Tekofsky","2022","blog","LessWrong","www.lesswrong.com/posts/cEQyKsreistXxFEeF/alignment-as-game-design",0,"",""],["All AGI safety questions welcome (especially basic ones) [July 2022]","plex and Robert Miles","2022","blog","LessWrong","www.lesswrong.com/posts/auPkxnLb3R9vXjEzo/all-agi-safety-questions-welcome-especially-basic-ones-july",0,"",""],["Do EA folks want AGI at all?","Noah Scales","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LfH9bywWRNSZozsx7/do-ea-folks-want-agi-at-all",0,"",""],["Does the idea of AGI that benevolently control us appeal to EA folks?","Noah Scales","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ipvSN3Hn5vejSi36q/does-the-idea-of-agi-that-benevolently-control-us-appeal-to",0,"",""],["How would a language model become goal-directed?","David Mears","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dgk2eLf8DLxEG6msd/how-would-a-language-model-become-goal-directed",0,"",""],["Perceiver AR: general-purpose, long-context autoregressive generation","Curtis Hawthorne and 14 others","2022","blog","deepmind.com","www.deepmind.com/blog/perceiver-ar-general-purpose-long-context-autoregressive-generation",0,"",""],["A note about differential technological development","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vQNJrJqebXEWjJfnz/a-note-about-differential-technological-development",0,"",""],["More to explore on 'Risks from Artificial Intelligence'","EA Handbook","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Cf6tNAhDbQFvAwbAg/more-to-explore-on-risks-from-artificial-intelligence",0,"",""],["Notes on Learning the Prior","Spencer Becker-Kahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ukidKsEio8hfB9uHT/notes-on-learning-the-prior",0,"",""],["Safety Implications of LeCun's path to machine intelligence","Ivan Vendrov","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GrbeyZzp6NwzSWpds/safety-implications-of-lecun-s-path-to-machine-intelligence",0,"",""],["What if we don't need a \"Hard Left Turn\" to reach AGI?","Eigengender","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JbScJgCDedXaBgyKC/what-if-we-don-t-need-a-hard-left-turn-to-reach-agi",0,"","governance forecasting"],["Circumventing interpretability: How to defeat mind-readers","Lee Sharkey","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EhAbh2pQoAXkm9yor/circumventing-interpretability-how-to-defeat-mind-readers",0,"","interpretability instrumental-convergence"],["Humans provide an untapped wealth of evidence about alignment","TurnTrout and Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CjFZeDD6iCnNubDoS/humans-provide-an-untapped-wealth-of-evidence-about",0,"",""],["It's OK not to go into AI (for students)","ruthgrace","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MK9AfXfkiz2mku6fv/it-s-ok-not-to-go-into-ai-for-students",0,"",""],["Resilience Via Fragmented Power","steve6320","2022","blog","EA Forum","forum.effectivealtruism.org/posts/A4KaDtjvGBANFa9Bb/resilience-via-fragmented-power",0,"",""],["Why policymakers should beware claims of new \"arms races\" (Bulletin of the Atomic Scientists)","christian.r","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BGDe6ZxfxyHTqNS5X/why-policymakers-should-beware-claims-of-new-arms-races",0,"","policy"],["Artificial Sandwiching: When can we test scalable alignment protocols without humans?","Sam Bowman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nekLYqbCEBDEfbLzF/artificial-sandwiching-when-can-we-test-scalable-alignment",0,"",""],["Deep learning curriculum for large language model alignment","Jacob_Hilton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5uNfgjaDAwkhcLJca/deep-learning-curriculum-for-large-language-model-alignment",0,"",""],["Goal Alignment Is Robust To the Sharp Left Turn","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/vix3K4grcHottqpEm/goal-alignment-is-robust-to-the-sharp-left-turn",0,"",""],["Making decisions using multiple worldviews","Richard_Ngo","2022","blog","LessWrong","www.lesswrong.com/posts/6QPFKHRsuY63cuJwh/making-decisions-using-multiple-worldviews",0,"","theory"],["MIRI Conversations: Technology Forecasting & Gradualism (Distillation)","TheMcDouglas","2022","blog","LessWrong","www.lesswrong.com/posts/DQMD5XZgBXaegRDQv/miri-conversations-technology-forecasting-and-gradualism",0,"","forecasting"],["Pile of Law and Law-Following AI","Cullen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JPdfFC3dM3Ksr4apo/pile-of-law-and-law-following-ai",0,"","governance policy"],["Searle vs Bostrom: crucial considerations for EA AI work?","Forumite","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xbF8fStkkRWbF9xg5/searle-vs-bostrom-crucial-considerations-for-ea-ai-work",0,"",""],["Slowing down AI progress is an underexplored alignment strategy","Michael Huang","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6LNvQYyNQpDQmnnux/slowing-down-ai-progress-is-an-underexplored-alignment",0,"","red-teaming governance"],["Acceptability Verification: A Research Agenda","David Udell and evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GeabLEXYP7oBMivmF/acceptability-verification-a-research-agenda",0,"",""],["AI ethics: the case for including animals (my first published paper, Peter Singer's first on AI)","Fai","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TjbkP3si2tjMkBMe3/ai-ethics-the-case-for-including-animals-my-first-published",0,"",""],["Mosaic and Palimpsests: Two Shapes of Research","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/4BpeHPXMjRzopgAZd/mosaic-and-palimpsests-two-shapes-of-research",0,"",""],["On how various plans miss the hard bits of the alignment challenge","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3pinFH3jerMzAvmza/on-how-various-plans-miss-the-hard-bits-of-the-alignment",0,"",""],["Recommendations for non-technical books on AI?","Joseph Lemien","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JbQRtNjiy9aq3Hv3o/recommendations-for-non-technical-books-on-ai",0,"",""],["Response to Blake Richards: AGI, generality, alignment, & loss functions","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rgPxEKFBLpLqJpMBM/response-to-blake-richards-agi-generality-alignment-and-loss",0,"",""],["What is wrong with this approach to corrigibility?","Rafael Cosman","2022","blog","LessWrong","www.lesswrong.com/posts/75hsimSa6BhT5pePH/what-is-wrong-with-this-approach-to-corrigibility",0,"",""],["EA for dumb people?","Olivia Addy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/x9Rn5SfapcbbZaZy9/ea-for-dumb-people",0,"","forecasting"],["Intuitive physics learning in a deep-learning model inspired by developmental psychology","Luis Piloto and 3 others","2022","blog","deepmind.com","www.deepmind.com/blog/intuitive-physics-learning-in-a-deep-learning-model-inspired-by-developmental-psychology",0,"",""],["Language Models (Mostly) Know What They Know","Saurav Kadavath and 35 others","2022","paper","arXiv preprint","arxiv.org/abs/2207.05221",0,"","evals"],["Hessian and Basin volume","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QPqztHpToij2nx7ET/hessian-and-basin-volume",0,"",""],["Immanuel Kant and the Decision Theory App Store","Daniel Kokotajlo","2022","blog","LessWrong","www.lesswrong.com/posts/byMNKEXBn4RTcaaa6/immanuel-kant-and-the-decision-theory-app-store",0,"","theory"],["Grouped Loss may disfavor discontinuous capabilities","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PmhTzHHFEem5hX79R/grouped-loss-may-disfavor-discontinuous-capabilities",0,"",""],["Making it harder for an AGI to \"trick\" us, with STVs","Tor Økland Barstad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xERh9dkBkHLHp7Lg6/making-it-harder-for-an-agi-to-trick-us-with-stvs",0,"","automated-alignment-research"],["Report from a civilizational observer on Earth","owencb","2022","blog","LessWrong","www.lesswrong.com/posts/KBjRoc7WccbMehnbN/report-from-a-civilizational-observer-on-earth",0,"","forecasting"],["Train first VS prune first in neural networks.","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PLqopCagHKo2EK5cE/train-first-vs-prune-first-in-neural-networks",0,"",""],["Visualizing Neural networks, how to blame the bias","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Bb33LG2YC3oTpBoGj/visualizing-neural-networks-how-to-blame-the-bias",0,"","interpretability"],["Reinforcement Learner Wireheading","Nate Showell","2022","blog","LessWrong","www.lesswrong.com/posts/4orwmWSosNyev4ByS/reinforcement-learner-wireheading",0,"","reward-hacking instrumental-convergence"],["Research Notes: What are we aligning for?","Shoshannah Tekofsky","2022","blog","LessWrong","www.lesswrong.com/posts/zpz929rJvJ8xCBiGZ/research-notes-what-are-we-aligning-for",0,"","instrumental-convergence"],["Human values & biases are inaccessible to the genome","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CQAMdzA4MZEhNRtTp/human-values-and-biases-are-inaccessible-to-the-genome",0,"",""],["Principles for Alignment/Agency Projects","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/A7GeRNLzuFnhvGGgb/principles-for-alignment-agency-projects",0,"",""],["Race Along Rashomon Ridge","Stephen Fowler and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hna2P8gcTyRgNDYBY/race-along-rashomon-ridge",0,"","interpretability"],["Safety considerations for online generative modeling","Sam Marks","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BMfNu82iunjqKyQA9/safety-considerations-for-online-generative-modeling",0,"",""],["Forecasting Through Fiction","Yitz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DhJhtxMX6SdYAsWiY/forecasting-through-fiction",0,"","red-teaming forecasting"],["How humanity would respond to slow takeoff, with takeaways from the entire COVID-19 pandemic","Noosphere89","2022","blog","LessWrong","www.lesswrong.com/posts/9r2P7kigDGZS8ysQz/how-humanity-would-respond-to-slow-takeoff-with-takeaways",0,"","forecasting"],["Inferring and Conveying Intentionality: Beyond Numerical Rewards to Logical Intentions","Susmit Jha and John Rushby","2022","paper","arXiv preprint","arxiv.org/abs/2207.05058",0,"","agents"],["Introducing the Fund for Alignment Research (We're Hiring!)","AdamGleave and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hGE3Pcc7qmK75bjhc/introducing-the-fund-for-alignment-research-we-re-hiring",0,"",""],["Introducing the Fund for Alignment Research (We're Hiring!)","AdamGleave and 3 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gNHjEmLeKM47FDdqM/introducing-the-fund-for-alignment-research-we-re-hiring-1",0,"",""],["Outer vs inner misalignment: three framings","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/poyshiMEhJsAuifKt/outer-vs-inner-misalignment-three-framings-1",0,"",""],["The History of AI Rights Research","Jamie Harris","2022","paper","arXiv preprint","arxiv.org/abs/2208.04714",0,"",""],["What work has been done on the post-AGI distribution of wealth?","levin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fPJfMWL5znqSzDSny/what-work-has-been-done-on-the-post-agi-distribution-of",0,"","forecasting"],["[AN #172] Sorry for the long hiatus!","Rohin Shah","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rxsdSgnZrgYWc2XAp/an-172-sorry-for-the-long-hiatus",0,"",""],["A central AI alignment problem: capabilities generalization, and the sharp left turn","Nate Soares","2022","blog","intelligence.org","intelligence.org/2022/07/04/a-central-ai-alignment-problem/",0,"",""],["Facilitator Help Wanted for Columbia EA AI Safety Groups","Berkan Ottlik","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vH9J7GEkYitWmjdGM/facilitator-help-wanted-for-columbia-ea-ai-safety-groups",0,"",""],["The curious case of Pretty Good human inner/outer alignment","PavleMiha","2022","blog","LessWrong","www.lesswrong.com/posts/A5XTgYmGnqEDnJdzJ/the-curious-case-of-pretty-good-human-inner-outer-alignment",0,"","robustness"],["A compressed take on recent disagreements","kman","2022","blog","LessWrong","www.lesswrong.com/posts/cwz5mRM5KvtdXftzk/a-compressed-take-on-recent-disagreements",0,"","forecasting"],["AI Forecasting: One Year In","jsteinhardt","2022","blog","LessWrong","www.lesswrong.com/posts/CJw2tNHaEimx6nwNy/ai-forecasting-one-year-in",0,"","forecasting"],["Benchmark for successful concept extrapolation/avoiding goal misgeneralization","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DiEWbwrChuzuhJhGr/benchmark-for-successful-concept-extrapolation-avoiding-goal",0,"","benchmarks"],["Future Matters #3: digital sentience, AGI ruin, and forecasting track records","Pablo and matthew.vandermerwe","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CictHfn8kdyupvpNK/future-matters-3-digital-sentience-agi-ruin-and-forecasting",0,"","forecasting"],["Human-centred mechanism design with Democratic AI","Raphael Koster and 10 others","2022","blog","deepmind.com","www.deepmind.com/blog/human-centred-mechanism-design-with-democratic-ai",0,"",""],["Is General Intelligence \"Compact\"?","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/J9XecqtiujawmDnmr/is-general-intelligence-compact",0,"","forecasting"],["New US Senate Bill on X-Risk Mitigation [Linkpost]","Evan R. Murphy","2022","blog","LessWrong","www.lesswrong.com/posts/bHBmmkdwsmjNwp94K/new-us-senate-bill-on-x-risk-mitigation-linkpost",0,"","governance"],["Please help us communicate AI xrisk. It could save the world.","otto.barten","2022","blog","LessWrong","www.lesswrong.com/posts/m7oMWaLQySRXcDznb/please-help-us-communicate-ai-xrisk-it-could-save-the-world",0,"","governance"],["Remaking EfficientZero (as best I can)","Hoagy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bPa6AzRgGZGmxbq6n/remaking-efficientzero-as-best-i-can",0,"",""],["Decision theory and dynamic inconsistency","paulfchristiano","2022","blog","LessWrong","www.lesswrong.com/posts/9W4TQvixiQjpZmzrx/decision-theory-and-dynamic-inconsistency",0,"","theory"],["Why AGI Timeline Research/Discourse Might Be Overrated","Miles_Brundage","2022","blog","EA Forum","forum.effectivealtruism.org/posts/SEqJoRL5Y8cypFasr/why-agi-timeline-research-discourse-might-be-overrated",0,"","forecasting"],["[Linkpost] Existential Risk Analysis in Empirical Research Papers","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5rNCGP8deEBjedCmH/linkpost-existential-risk-analysis-in-empirical-research",0,"",""],["Components of Strategic Clarity [Strategic Perspectives on Long-term AI Governance, #2]","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Bzezf2zmgBhtCD3Pb/components-of-strategic-clarity-strategic-perspectives-on",0,"","governance"],["Follow along with Columbia EA's Advanced AI Safety Fellowship!","RohanS","2022","blog","LessWrong","www.lesswrong.com/posts/mqZijPM2sEWJo9qMJ/follow-along-with-columbia-ea-s-advanced-ai-safety",0,"",""],["Naive Hypotheses on AI Alignment","Shoshannah Tekofsky","2022","blog","LessWrong","www.lesswrong.com/posts/ubdp8qAL8Gfki2pYo/naive-hypotheses-on-ai-alignment",0,"",""],["Research + Reality Graphing to Support AI Policy (and more): Summary of a Frozen Project","Harrison Durland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9RCFq976d9YXBbZyq/research-reality-graphing-to-support-ai-policy-and-more",0,"","governance policy"],["Strategic Perspectives on Transformative AI Governance: Introduction","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/isTXkKprgHh5j8WQr/strategic-perspectives-on-transformative-ai-governance",0,"","governance"],["The Linguistic Blind Spot of Value-Aligned Agency, Natural and Artificial","Travis LaCroix","2022","paper","arXiv preprint","arxiv.org/abs/2207.00868",0,"","agents"],["The Tree of Life: Stanford AI Alignment Theory of Change","Gabriel Mukobi","2022","blog","LessWrong","www.lesswrong.com/posts/zgJCSK5KdkiKDuuCw/the-tree-of-life-stanford-ai-alignment-theory-of-change",0,"",""],["When 2/3rds of the world goes against you","Jeffrey Kursonis","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6va2EfHkQ3bTmdDyn/when-2-3rds-of-the-world-goes-against-you",0,"","red-teaming"],["AI safety university groups: a promising opportunity to reduce existential risk","mic","2022","blog","LessWrong","www.lesswrong.com/posts/F6D5r2oq9CCnEBhWo/ai-safety-university-groups-a-promising-opportunity-to",0,"",""],["Artificial Intelligence, Morality, and Sentience (AIMS) Survey: 2021","Janet Pauketat and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DPDzKeQTyKEFDMwmg/artificial-intelligence-morality-and-sentience-aims-survey-1",0,"","governance forecasting"],["AXRP Episode 16 - Preparing for Debate AI with Geoffrey Irving","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bLr68nrLSwgzqLpzu/axrp-episode-16-preparing-for-debate-ai-with-geoffrey-irving",0,"",""],["generalized values: testing for patterns in computation","Tamsin Leake","2022","blog","carado.moe","carado.moe/generalized-values-testing-patterns.html",0,"",""],["Reframing the AI Risk","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/m3fyWQgCcFwro5KQh/reframing-the-ai-risk",0,"",""],["Safetywashing","Adam Scholl","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xhD6SHAAE9ghKZ9HS/safetywashing",0,"",""],["Trends in GPU price-performance","Marius Hobbhahn and Tamay","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/c6KFvQcZggQKZzxr9/trends-in-gpu-price-performance",0,"","forecasting scaling-laws"],["What Is The True Name of Modularity?","TheMcDouglas and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TTTHwLpcewGjQHWzh/what-is-the-true-name-of-modularity",0,"",""],["$500 bounty for alignment contest ideas","Akash and Olivia","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qNGv3N6feXYxjeYJb/usd500-bounty-for-alignment-contest-ideas-1",0,"",""],["(Even) More Early-Career EAs Should Try AI Safety Technical Research","levin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ycCBeG5SfApC3mcPQ/even-more-early-career-eas-should-try-ai-safety-technical",0,"","governance"],["Announcing the Harvard AI Safety Team","Xander Davies","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NvzeAtoynxGjDnWkp/announcing-the-harvard-ai-safety-team",0,"",""],["Forecasting Future World Events with Neural Networks","Andy Zou","2022","paper","arXiv preprint","arxiv.org/abs/2206.15474",0,"","policy forecasting"],["Formal Philosophy and Alignment Possible Projects","Whispermute","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fgAyy4gdDrbHFHjge/formal-philosophy-and-alignment-possible-projects",0,"",""],["Quick survey on AI alignment resources","frances_lorenz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/RecFq5M8NF8X98Gao/quick-survey-on-ai-alignment-resources",0,"",""],["The Track Record of Futurists Seems ... Fine","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/the-track-record-of-futurists-seems-fine/",0,"",""],["What are some current, already present challenges from AI?","nonzerosum","2022","blog","EA Forum","forum.effectivealtruism.org/posts/D3pyxpzxgff6H4sSv/what-are-some-current-already-present-challenges-from-ai",0,"",""],["Gradient hacking: definitions and examples","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EeAgytDZbDjRznPMA/gradient-hacking-definitions-and-examples",0,"",""],["Latent Adversarial Training","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/atBQ3NHyqnBadrsGP/latent-adversarial-training",0,"","deception"],["The inordinately slow spread of good AGI conversations in ML","RobBensinger","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qshYZfMDE5LETwbuf/the-inordinately-slow-spread-of-good-agi-conversations-in-ml",0,"","robustness"],["Will Capabilities Generalise More?","Ramana Kumar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cq5x4XDnLcBrYbb66/will-capabilities-generalise-more",0,"",""],["Doom doubts - is inner alignment a likely problem?","Crissman","2022","blog","LessWrong","www.lesswrong.com/posts/7ng5dKeGuJt4BmqgH/doom-doubts-is-inner-alignment-a-likely-problem",0,"",""],["Four reasons I find AI safety emotionally compelling","Kat Woods and Amber Dawn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nALZEFPcFd5sHJ2ew/four-reasons-i-find-ai-safety-emotionally-compelling",0,"",""],["Some alternative AI safety research projects","Michele Campolo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DypLJKRcQKt9hcpBP/some-alternative-ai-safety-research-projects",0,"",""],["What success looks like","Marius Hobbhahn and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/aKpqwtZN6ifAhqJYK/what-success-looks-like",0,"","governance"],["What success looks like","mariushobbhahn and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AuRBKFnjABa6c6GzC/what-success-looks-like",0,"","governance"],["Announcing Epoch: A research organization investigating the road to Transformative AI","Jsevillamol and 5 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AJ6GHm5n6fBRJbMhq/announcing-epoch-a-research-organization-investigating-the",0,"","governance forecasting"],["Announcing Epoch: A research organization investigating the road to Transformative AI","Jaime Sevilla and 5 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zqRDNChFburJMmpqK/announcing-epoch-a-research-organization-investigating-the",0,"","forecasting"],["Announcing the Inverse Scaling Prize ($250k Prize Pool)","Ethan Perez and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eqxqgFxymP8hXDTt5/announcing-the-inverse-scaling-prize-usd250k-prize-pool",0,"","scaling-laws"],["Auditing Visualizations: Transparency Methods Struggle to Detect Anomalous Behavior","Jean-Stanislas Denain and Jacob Steinhardt","2022","paper","arXiv preprint","arxiv.org/abs/2206.13498",0,"","interpretability evals"],["Deliberation Everywhere: Simple Examples","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5GzqD7fgtxjepPERp/deliberation-everywhere-simple-examples",0,"",""],["Deliberation, Reactions, and Control: Tentative Definitions and a Restatement of Instrumental Convergence","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/u4BaLRK6mJJcvycEk/deliberation-reactions-and-control-tentative-definitions-and",0,"","instrumental-convergence theory"],["Exploring Mild Behaviour in Embedded Agents","Megan Kinniment","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vNS4vGKGD6ygyz5ok/exploring-mild-behaviour-in-embedded-agents",0,"","agents theory"],["Military Artificial Intelligence as Contributor to Global Catastrophic Risk","MMMaas and Di Cooke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dCb8tWsAmbYPSiqYT/military-artificial-intelligence-as-contributor-to-global",0,"",""],["Parametrically Retargetable Decision-Makers Tend To Seek Power","Alexander Matt Turner and Prasad Tadepalli","2022","paper","arXiv preprint","arxiv.org/abs/2206.13477",0,"","power-seeking agents policy"],["Softmax Linear Units","Nelson Elhage and 32 others","2022","blog","transformer-circuits.pub","transformer-circuits.pub/2022/solu/index.html",0,"",""],["Training Trace Priors and Speed Priors","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9o2mjdo7eb7677EJS/training-trace-priors-and-speed-priors",0,"","deception"],["[LQ] Some Thoughts on Messaging Around AI Risk","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/eJrMaGZGut4Qaefsj/lq-some-thoughts-on-messaging-around-ai-risk",0,"",""],["AI-Written Critiques Help Humans Notice Flaws","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AHBejZBsaTR6dkRHs/ai-written-critiques-help-humans-notice-flaws",0,"",""],["Conditioning Generative Models","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nXeLPcT9uhfG3TMPS/conditioning-generative-models",0,"","deception"],["7 essays on Building a Better Future","Jamie_Harris","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ecfyChvYdRKXCTg5F/7-essays-on-building-a-better-future",0,"","forecasting"],["Multi-Modal and Multi-Factor Branching Time Active Inference","Théophile Champion and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.12503",0,"","robustness"],["Raphaël Millière on the Limits of Deep Learning and AI x-risk skepticism","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/TecosAABRtPhaNdrL/raphael-milliere-on-the-limits-of-deep-learning-and-ai-x",0,"",""],["Updated Deference is not a strong argument against the utility uncertainty approach to alignment","Ivan Vendrov","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ikYKHkffKNJvBygXG/updated-deference-is-not-a-strong-argument-against-the",0,"",""],["20 Critiques of AI Safety That I Found on Twitter","Daniel Kirmani","2022","blog","EA Forum","forum.effectivealtruism.org/posts/bb64TfCFbm8kNeGx4/20-critiques-of-ai-safety-that-i-found-on-twitter",0,"",""],["Formalizing the Problem of Side Effect Regularization","Alexander Matt Turner and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.11812",0,"","evals agents"],["Half-baked ideas thread (EA / AI Safety)","Aryeh Englander","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fuEHtmY9gvQQtdawL/half-baked-ideas-thread-ea-ai-safety",0,"",""],["Learning to play Minecraft with Video PreTraining","This was a large effort by a dedicated team. Each author made huge contributions on many fronts over long time periods. All members were full time on the project for over six months. BB and 4 others","2022","blog","openai.com","openai.com/research/vpt",0,"",""],["Loose thoughts on AGI risk","Yitz","2022","blog","LessWrong","www.lesswrong.com/posts/4RnpP3RP9HFkNzBdm/loose-thoughts-on-agi-risk",0,"","forecasting"],["Nonprofit Boards are Weird","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/nonprofit-boards-are-weird-2/",0,"",""],["On Avoiding Power-Seeking by Artificial Intelligence","Alexander Matt Turner","2022","paper","arXiv preprint","arxiv.org/abs/2206.11831",0,"","power-seeking agents"],["Confusion about neuroscience/cognitive science as a danger for AI Alignment","Samuel Nellessen","2022","blog","LessWrong","www.lesswrong.com/posts/DcW8ebBp38z7fAmyq/confusion-about-neuroscience-cognitive-science-as-a-danger",0,"",""],["Google's new text-to-image model - Parti, a demonstration of scaling benefits","Kayden","2022","blog","LessWrong","www.lesswrong.com/posts/tMr3HJwitJCbQ5HTc/google-s-new-text-to-image-model-parti-a-demonstration-of",0,"","scaling-laws"],["Reflection Mechanisms as an Alignment target: A survey","Marius Hobbhahn and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XyBWkoaqfnuEyNWXi/reflection-mechanisms-as-an-alignment-target-a-survey-1",0,"",""],["A Quick List of Some Problems in AI Alignment As A Field","NicholasKross","2022","blog","LessWrong","www.lesswrong.com/posts/yQJxi5mMZQopYXZk5/a-quick-list-of-some-problems-in-ai-alignment-as-a-field",0,"",""],["Getting from an unaligned AGI to an aligned AGI?","Tor Økland Barstad","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZmZBataeY58anJRBb/getting-from-an-unaligned-agi-to-an-aligned-agi",0,"","automated-alignment-research"],["Getting from an unaligned AGI to an aligned AGI?","Tor Økland Barstad","2022","blog","LessWrong","www.lesswrong.com/posts/ZmZBataeY58anJRBb/getting-from-an-unaligned-agi-to-an-aligned-agi",0,"","automated-alignment-research"],["Love and AI: Relational Brain/Mind Dynamics in AI Development","Jeffrey Kursonis","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MdfLn33GpNWGN7CSE/love-and-ai-relational-brain-mind-dynamics-in-ai-development",0,"","red-teaming"],["Technical AI safety in the United Arab Emirates","ea nyuad","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6XgtNEuzzKaWrvHHS/technical-ai-safety-in-the-united-arab-emirates",0,"",""],["The inordinately slow spread of good AGI conversations in ML","Rob Bensinger","2022","blog","LessWrong","www.lesswrong.com/posts/Rkxj7TFxhbm59AKJh/the-inordinately-slow-spread-of-good-agi-conversations-in-ml",0,"","robustness"],["Uncertainty Quantification for Competency Assessment of Autonomous Agents","Aastha Acharya and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.10553",0,"","agents forecasting"],["A Toy Model of Gradient Hacking","Oam Patel","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eDxhEDnKLfhtc28XK/a-toy-model-of-gradient-hacking",0,"",""],["BYOL-Explore: Exploration with Bootstrapped Prediction","Zhaohan Daniel Guo and 13 others","2022","blog","deepmind.com","www.deepmind.com/blog/byol-explore-exploration-with-bootstrapped-prediction",0,"",""],["Causal confusion as an argument against the scaling hypothesis","RobertKirk and David Scott Krueger (formerly: capybaralet)","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FZL4ftXvcuKmmobmj/causal-confusion-as-an-argument-against-the-scaling",0,"","scaling-laws"],["Digital Sentience Requires Solving the Boundary Problem","Andrés Gómez-Emilsson","2022","report","qri.org","qri.org/blog/digital-sentience",0,"",""],["How to become more agentic, by GPT-EA-Forum-v1","JoyOptimizer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kWbBwoqgSaadMtzf5/how-to-become-more-agentic-by-gpt-ea-forum-v1",0,"","agents"],["Key Papers in Language Model Safety","aogara","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zdA3ZpGZ5FxfaRgjb/key-papers-in-language-model-safety",0,"",""],["On corrigibility and its basin","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3Mwm7bpWgyvqwrMBT/on-corrigibility-and-its-basin",0,"",""],["recommending Hands and Cities","Tamsin Leake","2022","blog","carado.moe","carado.moe/hands-and-cities.html",0,"",""],["[linkpost] Christiano on agreement/disagreement with Yudkowsky's \"List of Lethalities\"","Owen Cotton-Barratt","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j7rj3ZyYbmacZXycn/linkpost-christiano-on-agreement-disagreement-with-yudkowsky",0,"",""],["Let's See You Write That Corrigibility Tag","Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AqsjZwxHNqH64C2b6/let-s-see-you-write-that-corrigibility-tag",0,"",""],["Modeling Transformative AI Risks (MTAIR) Project -- Summary Report","Sam Clarke and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.09360",0,"","agents"],["On Deference and Yudkowsky's AI Risk Estimates","bgarfinkel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NBgpPaz5vYe3tH4ga/on-deference-and-yudkowsky-s-ai-risk-estimates",0,"","red-teaming forecasting"],["Where I agree and disagree with Eliezer","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CoZhXrhpQxpy9xw9y/where-i-agree-and-disagree-with-eliezer",0,"",""],["Agent level parallelism","Johannes C. Mayer","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DC8a8aYXoHtc8bBaB/agent-level-parallelism-2",0,"","agents forecasting"],["Do yourself a FAVAR: security mindset","lukehmiles","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GBStFZinQYsyP8qHY/do-yourself-a-favar-security-mindset",0,"",""],["Scott Aaronson is joining OpenAI to work on AI safety","peterbarnett","2022","blog","LessWrong","www.lesswrong.com/posts/Tk5ovpucaqweCu4tu/scott-aaronson-is-joining-openai-to-work-on-ai-safety",0,"",""],["‘Force multipliers’ for EA research","Craig Drayton","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HcGnjibaHe6To9eaG/force-multipliers-for-ea-research",0,"",""],["AISC6: Research Summaries","Kristi Uustalu","2022","blog","aisafety.camp","aisafety.camp/2022/06/17/aisc6-research-summaries/",0,"",""],["anthropic reasoning coordination","Tamsin Leake","2022","blog","carado.moe","carado.moe/anthropic-reasoning-coordination.html",0,"",""],["Apply to the Machine Learning For Good bootcamp in France","Alexandre Variengien","2022","blog","EA Forum","forum.effectivealtruism.org/posts/4s7m4Eru29XKCL7xa/apply-to-the-machine-learning-for-good-bootcamp-in-france",0,"","robustness"],["Evolution through large models","OpenAI Research","2022","blog","openai.com","openai.com/research/evolution-through-large-models",0,"",""],["generalized computation interpretability","Tamsin Leake","2022","blog","carado.moe","carado.moe/generalized-computation-interpretability.html",0,"","interpretability"],["Pivotal outcomes and pivotal processes","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/etNJcXCsKC6izQQZj/pivotal-outcomes-and-pivotal-processes",0,"",""],["Pivotal outcomes and pivotal processes","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vqX25ML2vBN6cvmkx/pivotal-outcomes-and-pivotal-processes",0,"","governance"],["Quantifying General Intelligence","JasonBrown","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/a47XLgmX5ecWmxruY/quantifying-general-intelligence-2",0,"",""],["solonomonoff induction, time penalty, the universal program, and deism","Tamsin Leake","2022","blog","carado.moe","carado.moe/solomonoff-deism.html",0,"",""],["The Unified Theory of Normative Ethics","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/a3LncviZ6rkrTo8jJ/the-unified-theory-of-normative-ethics",0,"",""],["Unlocking High-Accuracy Differentially Private Image Classification through Scale","Soham De and 4 others","2022","blog","deepmind.com","www.deepmind.com/blog/unlocking-high-accuracy-differentially-private-image-classification-through-scale",0,"",""],["Value extrapolation vs Wireheading","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/snmwFzoDMQMhyTirN/value-extrapolation-vs-wireheading",0,"","reward-hacking"],["wrapper-minds are the enemy","nostalgebraist","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Mrz2srZWc7EzbADSo/wrapper-minds-are-the-enemy",0,"",""],["A transparency and interpretability tech tree","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nbq2bWLcYmSGup9aF/a-transparency-and-interpretability-tech-tree",0,"","interpretability"],["Breaking Down Goal-Directed Behaviour","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WhETfFgkfNSTShc4y/breaking-down-goal-directed-behaviour",0,"",""],["Characteristics of Harmful Text: Towards Rigorous Benchmarking of Language Models","Maribeth Rauh and 11 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.08325",0,"","evals benchmarks"],["Humans are very reliable agents","alyssavance","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/28zsuPaJpKAGSX4zq/humans-are-very-reliable-agents",0,"","agents"],["Interaction-Grounded Learning with Action-inclusive Feedback","Tengyang Xie and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.08364",0,"","agents policy"],["Refer the Cooperative AI Foundation’s New COO, Receive $5000","Lewis Hammond","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xLufaRBrDJuAXf3DB/refer-the-cooperative-ai-foundation-s-new-coo-receive",0,"",""],["Ten experiments in modularity, which we'd like you to run!","TheMcDouglas and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/99WtcMpsRqZcrocCd/ten-experiments-in-modularity-which-we-d-like-you-to-run",0,"",""],["Towards Gears-Level Understanding of Agency","Thane Ruthenis","2022","blog","LessWrong","www.lesswrong.com/posts/4pRPmFSfCLvKGEnFx/towards-gears-level-understanding-of-agency",0,"",""],["A central AI alignment problem: capabilities generalization, and the sharp left turn","So8res","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GNhMPAWcfBCASy8e6/a-central-ai-alignment-problem-capabilities-generalization",0,"",""],["Alignment Risk Doesn't Require Superintelligence","JustisMills","2022","blog","LessWrong","www.lesswrong.com/posts/EPEvBANNYN9rxRQFB/alignment-risk-doesn-t-require-superintelligence",0,"",""],["FYI: I’m working on a book about the threat of AGI/ASI for a general audience. I hope it will be of value to the cause and the community","Darren McKee","2022","blog","LessWrong","www.lesswrong.com/posts/E5MwmuuryBLF3wWRZ/fyi-i-m-working-on-a-book-about-the-threat-of-agi-asi-for-a",0,"","governance"],["What are all the AI Alignment and AI Safety Communication Hubs?","Gunnar_Zarncke","2022","blog","LessWrong","www.lesswrong.com/posts/vmLfa5PEcuAyZX3cC/what-are-all-the-ai-alignment-and-ai-safety-communication",0,"",""],["Blake Richards on Why he is Skeptical of Existential Risk from AI","Michaël Trazzi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BqokXcCQrvkk2BktH/blake-richards-on-why-he-is-skeptical-of-existential-risk",0,"",""],["Catholic theologians and priests on artificial intelligence","anonymous6","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dJy59zjes5qRKwc9S/catholic-theologians-and-priests-on-artificial-intelligence",0,"",""],["Expected impact of a career in AI safety under different opinions","Jordan Taylor","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tnxxy9SH9ebBkKDW7/expected-impact-of-a-career-in-ai-safety-under-different",0,"","forecasting"],["Investigating causal understanding in LLMs","Marius Hobbhahn and Tom Lieberum","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yZb5eFvDoaqB337X5/investigating-causal-understanding-in-llms",0,"",""],["Resources I send to AI researchers about AI safety","Vael Gates","2022","blog","LessWrong","www.lesswrong.com/posts/gdyfJE3noRFSs373q/resources-i-send-to-ai-researchers-about-ai-safety",0,"",""],["Slow motion videos as AI risk intuition pumps","Andrew_Critch","2022","blog","LessWrong","www.lesswrong.com/posts/Ccsx339LE9Jhoii9K/slow-motion-videos-as-ai-risk-intuition-pumps",0,"",""],["Steering AI to care for animals, and soon","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/35bfnGmsyrZkEnkLJ/steering-ai-to-care-for-animals-and-soon",0,"",""],["Vael Gates: Risks from Advanced AI (June 2022)","Vael Gates","2022","blog","EA Forum","forum.effectivealtruism.org/posts/q49obZkQujkYmnFWY/vael-gates-risks-from-advanced-ai-june-2022",0,"",""],["AI-written critiques help humans notice flaws","OpenAI Research","2022","blog","openai.com","openai.com/research/critiques",0,"",""],["Continuity Assumptions","Jan_Kulveit","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cHJxSJ4jBmBRGtbaE/continuity-assumptions",0,"","forecasting"],["Contra EY: Can AGI destroy us without trial & error?","Nikita Sokolsky","2022","blog","LessWrong","www.lesswrong.com/posts/qfDgEreMoSEtmLTws/contra-ey-can-agi-destroy-us-without-trial-and-error",0,"","forecasting"],["Towards Autonomous Grading In The Real World","Yakov Miron and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.06091",0,"","agents"],["Training Trace Priors","Adam Jermyn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hjrqXjEpaw9ogScPh/training-trace-priors",0,"","deception"],["What are current smaller problems related to top EA cause areas (eg deepfake policies for AI risk, ongoing covid variants for bio risk) and would it be beneficial for these small and not-catastrophic challenges to get more EA resources, as a way of developing capacity to prevent the catastrophic versions?","nonzerosum","2022","blog","EA Forum","forum.effectivealtruism.org/posts/huThsQPamDFJtG5tv/what-are-current-smaller-problems-related-to-top-ea-cause",0,"",""],["EA AI/Emerging Tech Orgs Should Be Involved with Patent Office Partnership","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/WoopYJtscbaodKQwh/ea-ai-emerging-tech-orgs-should-be-involved-with-patent",0,"",""],["Grokking “Semi-informative priors over AI timelines”","anson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kyG2thuWmi6bu3sKo/grokking-semi-informative-priors-over-ai-timelines",0,"","forecasting"],["Grokking “Semi-informative priors over AI timelines”","anson.ho","2022","blog","LessWrong","www.lesswrong.com/posts/Mj4CWRauhF3DzpLvu/grokking-semi-informative-priors-over-ai-timelines",0,"","forecasting"],["Why all the fuss about recursive self-improvement?","So8res","2022","blog","LessWrong","www.lesswrong.com/posts/8NKu9WES7KeKRWEKK/why-all-the-fuss-about-recursive-self-improvement",0,"","forecasting"],["AGI Ruin: A List of Lethalities","Eliezer Yudkowsky","2022","blog","intelligence.org","intelligence.org/2022/06/10/agi-ruin/",0,"",""],["AGI Safety Communications Initiative","Ines","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DS3frSuoNynzvjet4/agi-safety-communications-initiative",0,"",""],["ELK Proposal - Make the Reporter care about the Predictor’s beliefs","Adam Jermyn and Nicholas Schiefer","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XGbWaA3gbDphajMHm/elk-proposal-make-the-reporter-care-about-the-predictor-s",0,"","eliciting-latent-knowledge"],["Godzilla Strategies","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DwqgLXn5qYC7GqExF/godzilla-strategies",0,"","automated-alignment-research"],["Steganography and the CycleGAN - alignment failure case study","Jan Czechowski","2022","blog","LessWrong","www.lesswrong.com/posts/uutXLm2DRcCtFBZ2D/steganography-and-the-cyclegan-alignment-failure-case-study",0,"",""],["[linkpost] The final AI benchmark: BIG-bench","RomanS","2022","blog","LessWrong","www.lesswrong.com/posts/qjproXBGPQSAF9Hbd/linkpost-the-final-ai-benchmark-big-bench",0,"","benchmarks forecasting scaling-laws"],["AI Could Defeat All Of Us Combined","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6LTh4foNuC3NdtmZH/ai-could-defeat-all-of-us-combined",0,"","forecasting"],["AI Twitter accounts to follow?","Adrian Salustri","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FghLp7DJjLSFjzdQM/ai-twitter-accounts-to-follow",0,"",""],["Could Patent-Trolling delay AI timelines?","Pablo Repetto","2022","blog","LessWrong","www.lesswrong.com/posts/u8Ek5o5gyCErB8JHD/could-patent-trolling-delay-ai-timelines",0,"","governance forecasting"],["Digital people could make AI safer","GMcGowan","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ySCqffZTKtZp97JFB/digital-people-could-make-ai-safer",0,"",""],["Open Problems in AI X-Risk [PAIS #5]","Dan H and ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5HtDzRAk7ePWsiL2L/open-problems-in-ai-x-risk-pais-5",0,"",""],["Open Problems in AI X-Risk [PAIS #5]","ThomasW and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hNPCo4kScxccK9Ham/open-problems-in-ai-x-risk-pais-5",0,"",""],["Putting GPT-3's Creativity to the (Alternative Uses) Test","Claire Stevenson and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.08932",0,"",""],["why assume AGIs will optimize for fixed goals?","nostalgebraist","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dKTh9Td3KaJ8QW6gw/why-assume-agis-will-optimize-for-fixed-goals",0,"",""],["\"AI risk drone\"","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-risk-drone.html",0,"",""],["AI Could Defeat All Of Us Combined","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/ai-could-defeat-all-of-us-combined/",0,"",""],["diversity vs novelty","Tamsin Leake","2022","blog","carado.moe","carado.moe/diversity-novelty.html",0,"",""],["How Do Selection Theorems Relate To Interpretability?","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/A7QgKwWvAkuXonAy5/how-do-selection-theorems-relate-to-interpretability",0,"","interpretability"],["If no near-term alignment strategy, research should aim for the long-term","harsimony","2022","blog","LessWrong","www.lesswrong.com/posts/KuBcmj9pevPXNPB4C/if-no-near-term-alignment-strategy-research-should-aim-for",0,"",""],["If there was a millennium equivalent prize for AI alignment, what would the problems be?","Yair Halberstadt","2022","blog","LessWrong","www.lesswrong.com/posts/nEy2JDjvqE6o5E6kx/if-there-was-a-millennium-equivalent-prize-for-ai-alignment",0,"",""],["outer alignment: politics & philosophy","Tamsin Leake","2022","blog","carado.moe","carado.moe/outer-alignment-politics-philosophy.html",0,"",""],["Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","Chengyang Ying and 5 others","2022","paper","IJCAI 2022","arxiv.org/abs/2206.04436",0,"","agents policy"],["where are your alignment bits?","Tamsin Leake","2022","blog","carado.moe","carado.moe/alignment-bits.html",0,"",""],["You Only Get One Shot: an Intuition Pump for Embedded Agency","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HiufALieNbWHqR9en/you-only-get-one-shot-an-intuition-pump-for-embedded-agency",0,"","theory"],["Eliciting Latent Knowledge (ELK) - Distillation/Summary","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rxoBY9CMkqDsHt25t/eliciting-latent-knowledge-elk-distillation-summary",0,"","eliciting-latent-knowledge"],["Is the time crunch for AI Safety Movement Building now?","Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ZAP7gvPpD9YtoymxJ/is-the-time-crunch-for-ai-safety-movement-building-now",0,"",""],["Research Questions from Stained Glass Windows","StefanHex","2022","blog","LessWrong","www.lesswrong.com/posts/wx25pcqM6gvhPoJ4f/research-questions-from-stained-glass-windows",0,"","interpretability"],["Six Dimensions of Operational Adequacy in AGI Projects","Eliezer Yudkowsky","2022","blog","intelligence.org","intelligence.org/2022/06/07/six-dimensions-of-operational-adequacy-in-agi-projects/",0,"",""],["AGI Safety FAQ / all-dumb-questions-allowed thread","Aryeh Englander","2022","blog","LessWrong","www.lesswrong.com/posts/8c8AZq5hgifmnHKSN/agi-safety-faq-all-dumb-questions-allowed-thread",0,"",""],["CAISAR: A platform for Characterizing Artificial Intelligence Safety and Robustness","Julien Girard-Satabin and 4 others","2022","paper","AISafety, Jul 2022, Vienne, Austria","arxiv.org/abs/2206.03044",0,"","robustness"],["Imitating Past Successes can be Very Suboptimal","Benjamin Eysenbach and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.03378",0,"","policy"],["We will be around in 30 years","mukashi","2022","blog","LessWrong","www.lesswrong.com/posts/MLKmxZgtLYRH73um3/we-will-be-around-in-30-years",0,"","forecasting"],["Who models the models that model models? An exploration of GPT-3's in-context model fitting ability","Lovre","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/c2RzFadrxkzyRAFXa/who-models-the-models-that-model-models-an-exploration-of",0,"",""],["[Link] GCRI's Seth Baum reviews The Precipice","Aryeh Englander","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qxfmGBxAe5ZqdfDfv/link-gcri-s-seth-baum-reviews-the-precipice",0,"",""],["A descriptive, not prescriptive, overview of current AI Alignment Research","Jan and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FgjcHiWvADgsocE34/a-descriptive-not-prescriptive-overview-of-current-ai",0,"",""],["AGI Ruin: A List of Lethalities","EliezerYudkowsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zzFbZyGP6iz8jLe9n/agi-ruin-a-list-of-lethalities",0,"",""],["Epistemological Vigilance for Alignment","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/72scWeZRta2ApsKja/epistemological-vigilance-for-alignment",0,"",""],["Grokking “Forecasting TAI with biological anchors”","anson.ho","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wgio8E758y9XWsi8j/grokking-forecasting-tai-with-biological-anchors",0,"","forecasting"],["Grokking “Forecasting TAI with biological anchors”","anson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6RcicmJCvztarka8Y/grokking-forecasting-tai-with-biological-anchors",0,"","forecasting"],["Here are the finalists from FLI’s $100K Worldbuilding Contest","Jackson Wagner","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HEszzR4Am7PxN3hBG/here-are-the-finalists-from-fli-s-usd100k-worldbuilding",0,"","governance"],["Improving Model Understanding and Trust with Counterfactual Explanations of Model Confidence","Thao Le and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.02790",0,"","agents"],["Reading the ethicists 2: Hunting for AI alignment papers","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6DwprCdC7eErCRZkx/reading-the-ethicists-2-hunting-for-ai-alignment-papers",0,"",""],["Some ideas for follow-up projects to Redwood Research’s recent paper","JanBrauner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/7fBKErNKhtwB4nt4N/some-ideas-for-follow-up-projects-to-redwood-research-s",0,"","robustness"],["Why agents are powerful","Daniel Kokotajlo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/D5AzsRbRxZeqGuAZ4/why-agents-are-powerful",0,"","agents"],["AGI Ruin: A List of Lethalities","Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/uMQ3cqWDPHhjtiesc/agi-ruin-a-list-of-lethalities",0,"",""],["New cooperation mechanism - quadratic funding without a matching pool","Filip Sondej","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tXavWgk8Xp6Avg8No/new-cooperation-mechanism-quadratic-funding-without-a",0,"",""],["Announcing the Alignment of Complex Systems Research Group","Jan_Kulveit and technicalities","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/H5iGhDhQBtoDpCBZ2/announcing-the-alignment-of-complex-systems-research-group",0,"","agents"],["Deep Learning Systems Are Not Less Interpretable Than Logic/Probability/Etc","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/gebzzEwn2TaA6rGkc/deep-learning-systems-are-not-less-interpretable-than-logic",0,"","interpretability"],["How to pursue a career in technical AI alignment","CharlieRS","2022","blog","EA Forum","forum.effectivealtruism.org/posts/7WXPkpqKGKewAymJf/how-to-pursue-a-career-in-technical-ai-alignment",0,"",""],["How to pursue a career in technical AI alignment","charlie.rs","2022","blog","LessWrong","www.lesswrong.com/posts/iyKnennBbCvaWuKef/how-to-pursue-a-career-in-technical-ai-alignment",0,"",""],["Towards a Formalisation of Returns on Cognitive Reinvestment (Part 1)","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/cLyo7dKmimeXR3hAC/towards-a-formalisation-of-returns-on-cognitive-reinvestment",0,"","forecasting"],["Training a GPT model on EA texts: what data?","JoyOptimizer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AqfWhMvfiakEcpwfv/training-a-gpt-model-on-ea-texts-what-data",0,"",""],["[MLSN #4]: Many New Interpretability Papers, Virtual Logit Matching, Rationalization Helps Robustness","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/R39tGLeETfCZJ4FoE/mlsn-4-many-new-interpretability-papers-virtual-logit",0,"","interpretability robustness"],["Data collection for AI alignment - Career review","Benjamin Hilton and 80000_Hours","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dYf4w6qvidP7x5AND/data-collection-for-ai-alignment-career-review",0,"",""],["England & Wales & Windfalls","John Bridge","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DEFJkvzHeBdpmKQNR/england-and-wales-and-windfalls",0,"","governance policy"],["I'm trying out \"asteroid mindset\"","Alex_Altair","2022","blog","LessWrong","www.lesswrong.com/posts/PRMJCbBhsGgu5A6Ty/i-m-trying-out-asteroid-mindset",0,"","forecasting"],["Intergenerational trauma impeding cooperative existential safety efforts","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/SifFuesK7oc7DAMbw/intergenerational-trauma-impeding-cooperative-existential",0,"",""],["ML Safety Newsletter #4","Dan Hendrycks","2022","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-4",0,"",""],["Adversarial training, importance sampling, and anti-adversarial training for AI whistleblowing","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EFrsvnF6uZieZr3uG/adversarial-training-importance-sampling-and-anti",0,"",""],["Confused why a \"capabilities research is good for alignment progress\" position isn't discussed more","Kaj_Sotala","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EzAt4SbtQcXtDNhHK/confused-why-a-capabilities-research-is-good-for-alignment",0,"","robustness"],["Paradigms of AI alignment: components and enablers","Vika","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JC7aJZjt2WvxxffGz/paradigms-of-ai-alignment-components-and-enablers",0,"",""],["Paradigms of AI alignment: components and enablers","Victoria Krakovna","2022","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2022/06/02/paradigms-of-ai-alignment-components-and-enablers/",0,"",""],["Responsible/fair AI vs. beneficial/safe AI?","tae","2022","blog","EA Forum","forum.effectivealtruism.org/posts/usYMSj8S73tHFtJrC/responsible-fair-ai-vs-beneficial-safe-ai",0,"",""],["The Bio Anchors Forecast","Ansh Radhakrishnan","2022","blog","LessWrong","www.lesswrong.com/posts/FbP7EteJBCx8FpLFn/the-bio-anchors-forecast",0,"","forecasting"],["The prototypical catastrophic AI action is getting root access to its datacenter","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BAzCGCys4BkzGDCWR/the-prototypical-catastrophic-ai-action-is-getting-root",0,"",""],["Appendix to Bridging Demonstration","mako yass","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kB2DK6fpmbGCxe4ED/appendix-to-bridging-demonstration",0,"","governance"],["Contest: 250€ for translation of \"longtermism\" to German","constructive","2022","blog","EA Forum","forum.effectivealtruism.org/posts/iLbXBRwoc2ytpyqgq/contest-250eur-for-translation-of-longtermism-to-german",0,"",""],["Elucidating the Design Space of Diffusion-Based Generative Models","Tero Karras and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.00364",0,"","evals"],["HYCEDIS: HYbrid Confidence Engine for Deep Document Intelligence System","Bao-Sinh Nguyen and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.02628",0,"","evals"],["IDANI: Inference-time Domain Adaptation via Neuron-level Interventions","Omer Antverg and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2206.00259",0,"",""],["Machines vs Memes Part 3: Imitation and Memes","ceru23","2022","blog","LessWrong","www.lesswrong.com/posts/nbDFj4ZS6WSDKtSk4/machines-vs-memes-part-3-imitation-and-memes",0,"","instrumental-convergence"],["Advice on Pursuing Technical AI Safety Research","frances_lorenz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fRjj6nm9xbW4kFcTZ/advice-on-pursuing-technical-ai-safety-research",0,"",""],["Machines vs Memes Part 1: AI Alignment and Memetics","Harriet Farlow","2022","blog","LessWrong","www.lesswrong.com/posts/JLH6ido4qoBtYmnNR/machines-vs-memes-part-1-ai-alignment-and-memetics",0,"",""],["Machines vs. Memes 2: Memetically-Motivated Model Extensions","naterush","2022","blog","LessWrong","www.lesswrong.com/posts/gumkW3vy9mhjZriuc/machines-vs-memes-2-memetically-motivated-model-extensions",0,"",""],["Paper: Teaching GPT3 to express uncertainty in words","Owain_Evans","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vbfAwZqKs84agyGWC/paper-teaching-gpt3-to-express-uncertainty-in-words",0,"",""],["The Hard Intelligence Hypothesis and Its Bearing on Succession Induced Foom","DragonGod","2022","blog","LessWrong","www.lesswrong.com/posts/4iFSxvddsZCzBWtCo/the-hard-intelligence-hypothesis-and-its-bearing-on",0,"","forecasting"],["Two-Dimensional Quantum Material Identification via Self-Attention and Soft-labeling in Deep Learning","Xuan Bac Nguyen and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.15948",0,"","training-data"],["Which possible AI impacts should receive the most additional attention?","David Johnston","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HmDbjHBgopNDA6mrW/which-possible-ai-impacts-should-receive-the-most-additional",0,"",""],["Multi-Game Decision Transformers","Kuang-Huei Lee and 10 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.15241",0,"","evals agents"],["Perform Tractable Research While Avoiding Capabilities Externalities [Pragmatic AI Safety #4]","Dan H and ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dfRtxWcFDupfWpLQo/perform-tractable-research-while-avoiding-capabilities",0,"",""],["Perform Tractable Research While Avoiding Capabilities Externalities [Pragmatic AI Safety #4]","ThomasW and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/WmrCQTkTgDuk5RhCP/perform-tractable-research-while-avoiding-capabilities",0,"",""],["Pragmatic AI Safety","ThomasW","2022","report","forum.effectivealtruism.org","forum.effectivealtruism.org/s/8EqNwueP6iw2BQpNo",0,"",""],["Six Dimensions of Operational Adequacy in AGI Projects","Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/keiYkaeoLHoKK4LYA/six-dimensions-of-operational-adequacy-in-agi-projects",0,"","governance"],["Distilled - AGI Safety from First Principles","Harrison G","2022","blog","LessWrong","www.lesswrong.com/posts/2Enagkgxu49mRjDqe/distilled-agi-safety-from-first-principles",0,"",""],["Distributed Decisions","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/32sm7diYTky5KhF6w/distributed-decisions",0,"",""],["Multiple AIs in boxes, evaluating each other's alignment","Moebius314","2022","blog","LessWrong","www.lesswrong.com/posts/biskschef2zSNgKkz/multiple-ais-in-boxes-evaluating-each-other-s-alignment",0,"","evals"],["Reshaping the AI Industry","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mF8dkhZF9hAuLHXaD/reshaping-the-ai-industry",0,"","governance"],["The Problem With The Current State of AGI Definitions","Yitz","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EpR5yTZMaJkDz4hhs/the-problem-with-the-current-state-of-agi-definitions",0,"","forecasting"],["We should expect to worry more about speculative risks","bgarfinkel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/M68oj7fwXoPFJisap/we-should-expect-to-worry-more-about-speculative-risks",0,"",""],["concentric rings of illiberalism","Tamsin Leake","2022","blog","carado.moe","carado.moe/concentric-rings-illiberalism.html",0,"",""],["Teaching models to express their uncertainty in words","OpenAI Research","2022","blog","openai.com","openai.com/research/teaching-models-to-express-their-uncertainty-in-words",0,"",""],["Understanding Selection Theorems","adamk","2022","blog","LessWrong","www.lesswrong.com/posts/tdcLpkydLwcKwbKre/understanding-selection-theorems",0,"","agents theory"],["Croesus, Cerberus, and the magpies: a gentle introduction to Eliciting Latent Knowledge","Alexandre Variengien","2022","blog","LessWrong","www.lesswrong.com/posts/cv5xA2iSEnjz2Y9LF/croesus-cerberus-and-the-magpies-a-gentle-introduction-to",0,"","eliciting-latent-knowledge"],["Evaluating Multimodal Interactive Agents","Josh Abramson and 14 others","2022","blog","deepmind.com","www.deepmind.com/blog/evaluating-multimodal-interactive-agents",0,"","evals agents"],["GALOIS: Boosting Deep Reinforcement Learning via Generalizable Logic Synthesis","Yushi Cao and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.13728",0,"","interpretability evals policy"],["Infernal Corrigibility, Fiendishly Difficult","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/shb67DsGstZmvhiem/infernal-corrigibility-fiendishly-difficult",0,"",""],["Introducing spirit hazards","brb243","2022","blog","EA Forum","forum.effectivealtruism.org/posts/a4mFh3PySygwmWiAK/introducing-spirit-hazards",0,"",""],["Personalized Algorithmic Recourse with Preference Elicitation","Giovanni De Toni and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.13743",0,"","evals agents"],["plausible vs likely","Tamsin Leake","2022","blog","carado.moe","carado.moe/plausible-vs-likely.html",0,"",""],["say \"AI risk mitigation\" not \"alignment\"","Tamsin Leake","2022","blog","carado.moe","carado.moe/say-ai-risk-mitigation-not-alignment.html",0,"",""],["Where Utopias Go Wrong, or: The Four Little Planets","ExCeph","2022","blog","LessWrong","www.lesswrong.com/posts/jBNTf7o2R6bJjbJEk/where-utopias-go-wrong-or-the-four-little-planets",0,"",""],["A Story of AI Risk: InstructGPT-N","peterbarnett","2022","blog","LessWrong","www.lesswrong.com/posts/u8yT9bbabmdnpgDaQ/a-story-of-ai-risk-instructgpt-n",0,"",""],["Dynamic language understanding: adaptation to new knowledge in parametric and semi-parametric models","Elena Gribovskaya and 2 others","2022","blog","deepmind.com","www.deepmind.com/blog/dynamic-language-understanding-adaptation-to-new-knowledge-in-parametric-and-semi-parametric-models",0,"",""],["EA, Psychology & AI Safety Research","Sam Ellis","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fSDxnLcCn8h22gCYB/ea-psychology-and-ai-safety-research",0,"",""],["How Could AI Governance Go Wrong?","HaydnBelfield","2022","blog","EA Forum","forum.effectivealtruism.org/posts/7kj38wnMANwEAp6AT/how-could-ai-governance-go-wrong",0,"","governance"],["Infra-Bayesianism Distillation: Realizability and Decision Theory","Thomas Larsen","2022","blog","LessWrong","www.lesswrong.com/posts/DMoiZDYZzqknfvoHh/infra-bayesianism-distillation-realizability-and-decision",0,"","theory"],["The pointers problem, distilled","Nina Rimsky","2022","blog","LessWrong","www.lesswrong.com/posts/r4ksbGjoighPsXyXi/the-pointers-problem-distilled",0,"",""],["A Human-Centric Assessment Framework for AI","Sascha Saralajew and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.12749",0,"",""],["autonomy: the missing AGI ingredient?","nostalgebraist","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HSETWwdJnb45jsvT8/autonomy-the-missing-agi-ingredient",0,"",""],["RL with KL penalties is better seen as Bayesian inference","Tomek Korbak and Ethan Perez","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eoHbneGvqDu25Hasc/rl-with-kl-penalties-is-better-seen-as-bayesian-inference",0,"",""],["The \"Measuring Stick of Utility\" Problem","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/73pTioGZKNcfQmvGF/the-measuring-stick-of-utility-problem",0,"",""],["Complex Systems for AI Safety [Pragmatic AI Safety #3]","Dan H and ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/n767Q8HqbrteaPA25/complex-systems-for-ai-safety-pragmatic-ai-safety-3",0,"",""],["Complex Systems for AI Safety [Pragmatic AI Safety #3]","ThomasW and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eHYxg7cFxqQPGo7hD/complex-systems-for-ai-safety-pragmatic-ai-safety-3",0,"",""],["Explaining inner alignment to myself","Jeremy Gillen","2022","blog","LessWrong","www.lesswrong.com/posts/HtEffpHcLxppLN6dL/explaining-inner-alignment-to-myself",0,"",""],["The No Free Lunch theorems and their Razor","Adrià Garriga-alonso","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/sdrCBWpqyvNBJSZH5/the-no-free-lunch-theorems-and-their-razor",0,"",""],["2022 Uehiro Lectures: Ethics and Artificial Intelligence","Peter Railton","2022","report","practicalethics.ox.ac.uk","www.practicalethics.ox.ac.uk/uehiro-lectures-2022",0,"",""],["AXRP Episode 15 - Natural Abstractions with John Wentworth","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/L896Fp8hLSbh8Ryei/axrp-episode-15-natural-abstractions-with-john-wentworth",0,"","agents theory"],["Bits of Optimization Can Only Be Lost Over A Distance","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tpHB69eXorChEsix3/bits-of-optimization-can-only-be-lost-over-a-distance",0,"",""],["Gradations of Agency","Daniel Kokotajlo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ndcqdTxkMnFF7gQFh/gradations-of-agency-1",0,"","agents"],["RL with KL penalties is better viewed as Bayesian inference","Tomasz Korbak","2022","paper","arXiv preprint","arxiv.org/abs/2205.11275",0,"","policy"],["The Windfall Clause has a remedies problem","John Bridge","2022","blog","EA Forum","forum.effectivealtruism.org/posts/wBzfLyfJFfocmdrwL/the-windfall-clause-has-a-remedies-problem",0,"","red-teaming governance policy forecasting"],["Why I'm Worried About AI","peterbarnett","2022","blog","LessWrong","www.lesswrong.com/posts/k8hvGAJWSKAeHwpnJ/why-i-m-worried-about-ai",0,"","deception"],["X-Risk Motivations for Safety Research Directions","Dan Hendrycks and 3 others","2022","report","docs.google.com","docs.google.com/document/d/1PXwjSbh-g1U1JEXhf55C7YsqD5qKNdBPQGgQ7W0Zm1A/edit?usp=sharing",0,"",""],["Adversarial attacks and optimal control","Jan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/KtCJNw93KHg7MSSvw/adversarial-attacks-and-optimal-control",0,"","robustness"],["implementing the platonic realm","Tamsin Leake","2022","blog","carado.moe","carado.moe/implementing-the-platonic-realm.html",0,"",""],["Responsible Artificial Intelligence -- from Principles to Practice","Virginia Dignum","2022","paper","arXiv preprint","arxiv.org/abs/2205.10785",0,"","interpretability"],["SERI ML application deadline is extended until May 22.","Viktoria Malyasova","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gxDgnxDW2KoayK4Bb/seri-ml-application-deadline-is-extended-until-may-22",0,"",""],["[Short version] Information Loss --> Basin flatness","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SZM32BdvYgrsBfYnw/short-version-information-loss-greater-than-basin-flatness",0,"",""],["AI boxing could be easy","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-boxing-easy.html",0,"",""],["Clarifying what ELK is trying to achieve","Simon Skade","2022","blog","LessWrong","www.lesswrong.com/posts/8xCtJHAbzyA2oA6J4/clarifying-what-elk-is-trying-to-achieve",0,"","eliciting-latent-knowledge"],["Information Loss --> Basin flatness","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wPudaEemohdYPmsye/information-loss-greater-than-basin-flatness",0,"",""],["Scaling Laws and Interpretability of Learning from Repeated Data","Danny Hernandez and 17 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.10487",0,"","interpretability mechanistic-interpretability scaling-laws"],["Exploring the Trade-off between Plausibility, Change Intensity and Adversarial Power in Counterfactual Explanations using Multi-objective Optimization","Javier Del Ser and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.10232",0,"","interpretability"],["How RL Agents Behave When Their Actions Are Modified? [Distillation post]","PabloAMC","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FeY4tXMYdTQSM4go3/how-rl-agents-behave-when-their-actions-are-modified",0,"","agents"],["Are you really in a race? The Cautionary Tales of Szilárd and Ellsberg","HaydnBelfield","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cXBznkfoPJAjacFoT/are-you-really-in-a-race-the-cautionary-tales-of-szilard-and",0,"","red-teaming governance"],["[Fiction] Improved Governance on the Critical Path to AI Alignment by 2045.","Jackson Wagner","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LLfaikCmysmdxussN/fiction-improved-governance-on-the-critical-path-to-ai",0,"","governance forecasting"],["[Link] Reading the ethicists: A review of articles on AI in the journal Science and Engineering Ethics","Charlie Steiner","2022","blog","EA Forum","forum.effectivealtruism.org/posts/oxfttm6b9zk6TbZKJ/link-reading-the-ethicists-a-review-of-articles-on-ai-in-the",0,"",""],["A bridge to Dath Ilan? Improved governance on the critical path to AI alignment.","Jackson Wagner","2022","blog","LessWrong","www.lesswrong.com/posts/qo2hqf2ha7rfgCdjY/a-bridge-to-dath-ilan-improved-governance-on-the-critical",0,"","governance"],["Gato's Generalisation: Predictions and Experiments I'd Like to See","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LnnMPNHEpqtaqonCM/gato-s-generalisation-predictions-and-experiments-i-d-like",0,"","forecasting"],["generalized adding reality layers","Tamsin Leake","2022","blog","carado.moe","carado.moe/generalized-adding-reality-layers.html",0,"",""],["How to get into AI safety research","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WFnyxSL543c9bxGMm/how-to-get-into-ai-safety-research",0,"",""],["Maxent and Abstractions: Current Best Arguments","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cqdDGuTs2NamtEhBW/maxent-and-abstractions-current-best-arguments",0,"",""],["Mimicking Behaviors in Separated Domains","Giuseppe De Giacomo and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.09201",0,"","agents"],["predictablizing ethic deduplication","Tamsin Leake","2022","blog","carado.moe","carado.moe/predictablizing-ethic-deduplication.html",0,"",""],["We have achieved Noob Gains in AI","phdead","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QBgFnBMJpmGkE5ioc/we-have-achieved-noob-gains-in-ai",0,"",""],["[Intro to brain-like-AGI safety] 15. Conclusion: Open problems, how to help, AMA","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tj8AC3vhTnBywdZoA/intro-to-brain-like-agi-safety-15-conclusion-open-problems-1",0,"",""],["Actionable-guidance and roadmap recommendations for the NIST AI Risk Management Framework","Dan H and Tony Barrett","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JNqXyEuKM4wbFZzpL/actionable-guidance-and-roadmap-recommendations-for-the-nist-1",0,"","governance"],["Actionable-guidance and roadmap recommendations for the NIST AI Risk Management Framework","Tony Barrett and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EEYvKn7gjps5wjFFq/actionable-guidance-and-roadmap-recommendations-for-the-nist",0,"","evals governance policy"],["LW4EA: Some cruxes on impactful alternatives to AI policy work","Jeremy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MtZkQzE5yuiJrw9wd/lw4ea-some-cruxes-on-impactful-alternatives-to-ai-policy",0,"","policy"],["We Ran an AI Timelines Retreat","Lenny McCline","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EZQQmhMsa36zwPeGB/we-ran-an-ai-timelines-retreat",0,"","forecasting"],["“Intro to brain-like-AGI safety” series—just finished!","Steven Byrnes","2022","blog","EA Forum","forum.effectivealtruism.org/posts/p5hfDvD59tidpLyi8/intro-to-brain-like-agi-safety-series-just-finished",0,"",""],["AGI Risk: How to internationally regulate industries in non-democracies","Timothy_Liptrot","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Zu2CTGP5xDR9nusoG/agi-risk-how-to-internationally-regulate-industries-in-non",0,"","evals governance policy"],["DeepMind’s generalist AI, Gato: A non-technical explainer","frances_lorenz and 2 others","2022","blog","LessWrong","www.lesswrong.com/posts/756HbyEBkL3xLSe7b/deepmind-s-generalist-ai-gato-a-non-technical-explainer",0,"","governance"],["EleutherAI Alignment 101","Richard Ngo","2022","report","drive.google.com","drive.google.com/file/d/1pkMNjaJsgogoqfGcaUHzmILhn5PsHSQt/view?usp=share_link",0,"",""],["Emergent Bartering Behaviour in Multi-Agent Reinforcement Learning","Mike Johanson and 3 others","2022","blog","deepmind.com","www.deepmind.com/blog/emergent-bartering-behaviour-in-multi-agent-reinforcement-learning",0,"","agents"],["How Different Groups Prioritize Ethical Values for Responsible AI","Maurice Jakesch and 3 others","2022","paper","2022 ACM Conference on Fairness, Accountability, and Transparency\n  (FAccT '22), June 21-24, 2022, Seoul, Republic of Korea","arxiv.org/abs/2205.07722",0,"",""],["Optimization at a Distance","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/d2n74bwham8motxyX/optimization-at-a-distance",0,"","theory"],["Proxy misspecification and the capabilities vs. value learning race","Sam Marks","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tWpgtjRm9qwzxAZEi/proxy-misspecification-and-the-capabilities-vs-value",0,"","goodharts-law"],["To what extent is your AGI timeline bimodal or otherwise \"bumpy\"?","jchan","2022","blog","LessWrong","www.lesswrong.com/posts/6tp4YAXjWLM3HocW7/to-what-extent-is-your-agi-timeline-bimodal-or-otherwise",0,"","forecasting"],["Gato as the Dawn of Early AGI","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/TwfWTLhQZgy2oFwK3/gato-as-the-dawn-of-early-agi",0,"","forecasting"],["smaller X-risk","Tamsin Leake","2022","blog","carado.moe","carado.moe/smaller-x-risk.html",0,"",""],["The AI Countdown Clock","River Lewis","2022","blog","LessWrong","www.lesswrong.com/posts/AhhdyxiAG6669BxLe/the-ai-countdown-clock",0,"","forecasting"],["What does the Project Management role look like in AI safety?","gvst","2022","blog","EA Forum","forum.effectivealtruism.org/posts/yTyooCqTfWSCpTzPB/what-does-the-project-management-role-look-like-in-ai-safety",0,"",""],["\"Tech company singularities\", and steering them to reduce x-risk","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/KopQknZEtjZdoGorT/tech-company-singularities-and-steering-them-to-reduce-x",0,"",""],["\"Tech company singularities\", and steering them to reduce x-risk","Andrew_Critch","2022","blog","LessWrong","www.lesswrong.com/posts/ezGYBHTxiRgmMRpWK/tech-company-singularities-and-steering-them-to-reduce-x",0,"","forecasting"],["Against Time in Agent Models","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HCibBn3ZCZRwMwNEE/against-time-in-agent-models",0,"","agents"],["Agency As a Natural Abstraction","Thane Ruthenis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/n2urKnXbevj2ryvGY/agency-as-a-natural-abstraction",0,"",""],["Alignment as Constraints","Logan Riggs","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/7GGRmAyMzqzidmBbi/alignment-as-constraints",0,"",""],["An observation about Hubinger et al.'s framework for learned optimization","Spencer Becker-Kahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rAhJrdxjsXcngn3ip/an-observation-about-hubinger-et-al-s-framework-for-learned",0,"",""],["Clarifying the confusion around inner alignment","Rauno Arike","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xdtNd8xCdzpgfnGme/clarifying-the-confusion-around-inner-alignment",0,"",""],["cognitive biases regarding the evaluation of AI risk when doing AI capabilities work","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-capability-risk-biases.html",0,"","evals"],["DeepMind is hiring for the Scalable Alignment and Alignment Teams","Rohin Shah and Geoffrey Irving","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/nzmCvRvPm4xJuqztv/deepmind-is-hiring-for-the-scalable-alignment-and-alignment",0,"",""],["Fermi estimation of the impact you might have working on AI safety","frib","2022","blog","EA Forum","forum.effectivealtruism.org/posts/widWpunQMfuNTCYE3/fermi-estimation-of-the-impact-you-might-have-working-on-ai",0,"",""],["Frame for Take-Off Speeds to inform compute governance & scaling alignment","Logan Riggs","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XFL3vaA69mHxATWM7/frame-for-take-off-speeds-to-inform-compute-governance-and",0,"","governance compute-governance forecasting"],["I'm interviewing Max Tegmark about AI safety and more. What shouId I ask him?","Robert_Wiblin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AQrthFKWgJvMWw5JB/i-m-interviewing-max-tegmark-about-ai-safety-and-more-what",0,"",""],["Thoughts on AI Safety Camp","Charlie Steiner","2022","blog","LessWrong","www.lesswrong.com/posts/3kErRpEprB8iJvnNq/thoughts-on-ai-safety-camp",0,"",""],["A Generalist Agent","Scott Reed and 19 others","2022","blog","deepmind.com","www.deepmind.com/blog/a-generalist-agent",0,"","agents"],["A tentative dialogue with a Friendly-boxed-super-AGI on brain uploads","Ramiro","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QZCuGZczixXXJeEyw/a-tentative-dialogue-with-a-friendly-boxed-super-agi-on",0,"",""],["Deepmind's Gato: Generalist Agent","Daniel Kokotajlo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xxvKhjpcTAJwvtbWM/deepmind-s-gato-generalist-agent",0,"","agents forecasting"],["Interpretability’s Alignment-Solving Potential: Analysis of 7 Scenarios","Evan R. Murphy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FrFZjkdRsmsbnQEm8/interpretability-s-alignment-solving-potential-analysis-of-7",0,"","scalable-oversight interpretability eliciting-latent-knowledge"],["Introduction to the sequence: Interpretability Research for the Most Important Century","Evan R. Murphy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/MygKP4iwdRL24eNsY/introduction-to-the-sequence-interpretability-research-for-1",0,"","interpretability"],["life refocus","Tamsin Leake","2022","blog","carado.moe","carado.moe/life-refocus.html",0,"",""],["New series of posts answering one of Holden's \"Important, actionable research questions\"","Evan R. Murphy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/L9ogLxNWuCbPM9AsP/new-series-of-posts-answering-one-of-holden-s-important",0,"","interpretability"],["[Intro to brain-like-AGI safety] 14. Controlled AGI","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QpHewJvZJFaQYuLwH/intro-to-brain-like-agi-safety-14-controlled-agi",0,"",""],["[Intro to brain-like-AGI safety] 14. Controlled AGI","Steven Byrnes","2022","blog","LessWrong","www.lesswrong.com/posts/QpHewJvZJFaQYuLwH/intro-to-brain-like-agi-safety-14-controlled-agi",0,"",""],["AI risk plans","Tamsin Leake","2022","blog","carado.moe","carado.moe/ai-risk-plans.html",0,"",""],["hope for infinite compute","Tamsin Leake","2022","blog","carado.moe","carado.moe/hope-infinite-compute.html",0,"",""],["AI safety should be made more accessible using non text-based media","Massimog","2022","blog","LessWrong","www.lesswrong.com/posts/bDG4swEX6smpRZvsX/ai-safety-should-be-made-more-accessible-using-non-text",0,"","governance"],["Dath Ilani Rule of Law","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/E7XGYmvRSigjHX8uz/dath-ilani-rule-of-law",0,"","theory"],["Rabbits, robots and resurrection","Patrick Wilson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/h2EaaDchr9QYuKz9z/rabbits-robots-and-resurrection",0,"",""],["The limits of AI safety via debate","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kguLeJTt6LnGuYX4E/the-limits-of-ai-safety-via-debate",0,"","debate"],["A Bird's Eye View of the ML Field [Pragmatic AI Safety #2]","Dan H and ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AtfQFj8umeyBBkkxa/a-bird-s-eye-view-of-the-ml-field-pragmatic-ai-safety-2",0,"",""],["A Bird's Eye View of the ML Field [Pragmatic AI Safety #2]","ThomasW and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/PFxmd5bf7nqGNLYCg/a-bird-s-eye-view-of-the-ml-field-pragmatic-ai-safety-2",0,"","forecasting"],["AI Alignment YouTube Playlists","jacquesthibs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CLN2F6HhhSYhwixAC/ai-alignment-youtube-playlists",0,"",""],["AI Alternative Futures: Exploratory Scenario Mapping for Artificial Intelligence Risk - Request for Participation [Linkpost]","Kiliank","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JSko4DZsppThDN7iP/ai-alternative-futures-exploratory-scenario-mapping-for",0,"","instrumental-convergence governance forecasting"],["Aligned with Whom? Direct and social goals for AI systems","Anton Korinek and Avital Balwit","2022","paper","arXiv preprint","arxiv.org/abs/2205.04279",0,"","governance"],["Conditions for mathematical equivalence of Stochastic Gradient Descent and Natural Selection","Oliver Sourbut","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5XbBm6gkuSdMJy9DT/conditions-for-mathematical-equivalence-of-stochastic",0,"",""],["Introduction to Pragmatic AI Safety [Pragmatic AI Safety #1]","Dan H and ThomasW","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bffA9WC9nEJhtagQi/introduction-to-pragmatic-ai-safety-pragmatic-ai-safety-1",0,"",""],["Introduction to Pragmatic AI Safety [Pragmatic AI Safety #1]","ThomasW and Dan H","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MskKEsj8nWREoMjQK/introduction-to-pragmatic-ai-safety-pragmatic-ai-safety-1",0,"","red-teaming"],["Jobs: Help scale up LM alignment research at NYU","Sam Bowman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5MHxjWgwWEoMrPXj8/jobs-help-scale-up-lm-alignment-research-at-nyu",0,"",""],["Student project for engaging with AI alignment","Per Ivar Friborg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cMvxw4ehHJy2vYJDA/student-project-for-engaging-with-ai-alignment",0,"",""],["Transcripts of interviews with AI researchers","Vael Gates","2022","blog","LessWrong","www.lesswrong.com/posts/LfHWhcfK92qh2nwku/transcripts-of-interviews-with-ai-researchers",0,"",""],["Updating Utility Functions","JustinShovelain and Joar Skalse","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NjYdGP59Krhie4WBp/updating-utility-functions",0,"",""],["When is AI safety research harmful?","Nathan_Barnard","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kcopiC5G4nagd4ndd/when-is-ai-safety-research-harmful",0,"","governance"],["A Survey on AI Sustainability: Emerging Trends on Learning Algorithms and Research Challenges","Zhenghua Chen and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.03824",0,"","robustness"],["Algorithmic formalization of FDT?","shminux","2022","blog","LessWrong","www.lesswrong.com/posts/Nw5MwgJBGXSWqaKag/algorithmic-formalization-of-fdt",0,"","theory"],["Elementary Infra-Bayesianism","Jan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9uj2Mto9CNdWZudyq/elementary-infra-bayesianism",0,"",""],["README-by Vael Gates-date 20220509","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1haxtGQ8aigy9A-7SpbXTqGCg8tSF9D2a/view?usp=share_link",0,"",""],["Video and Transcript of Presentation on Existential Risk from Power-Seeking AI","Joe_Carlsmith","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ChuABPEXmRumcJY57/video-and-transcript-of-presentation-on-existential-risk",0,"","power-seeking"],["Video and Transcript of Presentation on Existential Risk from Power-Seeking AI","Joe Carlsmith","2022","blog","LessWrong","www.lesswrong.com/posts/76etTtAiKtZGGzkmi/video-and-transcript-of-presentation-on-existential-risk",0,"","power-seeking"],["What are the coolest topics in AI safety, to a hopelessly pure mathematician?","Jenny K E","2022","blog","EA Forum","forum.effectivealtruism.org/posts/d7fJLQz2QaDNbbWxJ/what-are-the-coolest-topics-in-ai-safety-to-a-hopelessly",0,"",""],["What does Functional Decision Theory say to do in imperfect Newcomb situations?","Daniel_Eth","2022","blog","LessWrong","www.lesswrong.com/posts/NGj4KrTYsyH57SxYC/what-does-functional-decision-theory-say-to-do-in-imperfect",0,"","theory"],["Active offline policy selection","Yutian Chen and 7 others","2022","blog","deepmind.com","www.deepmind.com/blog/active-offline-policy-selection",0,"","policy"],["Apply to the second iteration of the ML for Alignment Bootcamp (MLAB 2) in Berkeley [Aug 15 - Fri Sept 2]","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3ouxBRRzjxarTukMW/apply-to-the-second-iteration-of-the-ml-for-alignment",0,"",""],["Apply to the second ML for Alignment Bootcamp (MLAB 2) in Berkeley [Aug 15 - Fri Sept 2]","Buck and Max Nadeau","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vvocfhQ7bcBR4FLBx/apply-to-the-second-ml-for-alignment-bootcamp-mlab-2-in",0,"","robustness"],["Open Problems in Negative Side Effect Minimization","Fabian Schimpf and Lukas Fluri","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pnAxcABq9GBDG5BNW/open-problems-in-negative-side-effect-minimization",0,"",""],["The case for becoming a black-box investigator of language models","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yGaw4NqRha8hgx5ny/the-case-for-becoming-a-black-box-investigator-of-language",0,"","interpretability"],["A Deep Reinforcement Learning Framework for Rapid Diagnosis of Whole Slide Pathological Images","Tingting Zheng and 9 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.02850",0,"","agents"],["High-stakes alignment via adversarial training [Redwood Research report]","dmz and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/A9tJFJY7DsGTFKKkh/high-stakes-alignment-via-adversarial-training-redwood",0,"","robustness"],["How to use the Forum (intro)","Lizka","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ht2dScQTpeBXB6uMb/how-to-use-the-forum-intro",0,"",""],["Messy personal stuff that affected my cause prioritization (or: how I started to care about AI safety)","Julia_Wise","2022","blog","EA Forum","forum.effectivealtruism.org/posts/mZ4ctSAEMgWj6DAwt/messy-personal-stuff-that-affected-my-cause-prioritization",0,"",""],["The AI Messiah","ryancbriggs","2022","blog","EA Forum","forum.effectivealtruism.org/posts/r72wjMns9wyaAhWhc/the-ai-messiah",0,"",""],["Ethan Caballero-by The Inside View-date 20220505","Ethan","2022","report","drive.google.com","drive.google.com/file/d/1RDr8WcSNLKZOa78W6IfI3CJZ6nVs5jEJ/view?usp=share_link",0,"",""],["Introducing the ML Safety Scholars Program","Dan H and 5 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CphfDP4ynz3QQ4AKY/introducing-the-ml-safety-scholars-program",0,"",""],["Introducing the ML Safety Scholars Program","ThomasW and 5 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9RYvJu2iNJMXgWCBn/introducing-the-ml-safety-scholars-program",0,"",""],["Adversarial Training for High-Stakes Reliability","Authors: Daniel M. Ziegler and 11 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.01663",0,"","robustness"],["Is evolutionary influence the mesa objective that we're interested in?","David Johnston","2022","blog","LessWrong","www.lesswrong.com/posts/me34KqMLwJNYAZKbs/is-evolutionary-influence-the-mesa-objective-that-we-re",0,"",""],["Information security considerations for AI and the long term future","Jeffrey Ladish and lennart","2022","blog","EA Forum","forum.effectivealtruism.org/posts/WqQDCCLWbYfFRwubf/information-security-considerations-for-ai-and-the-long-term",0,"","governance"],["My thoughts on nanotechnology strategy research as an EA cause area","Ben Snodin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/oqBJk2Ae3RBegtFfn/my-thoughts-on-nanotechnology-strategy-research-as-an-ea",0,"","forecasting"],["The AI Index 2022 Annual Report","Daniel Zhang and 12 others","2022","paper","arXiv preprint","arxiv.org/abs/2205.03468",0,"","policy"],["What are the best journals to publish AI governance papers in?","CaroJ","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JNCmoe3fno2mhbb4o/what-are-the-best-journals-to-publish-ai-governance-papers",0,"","governance"],["A tale of 2.5 orthogonality theses","Arepo","2022","blog","EA Forum","forum.effectivealtruism.org/posts/kCAcrjvXDt2evMpBz/a-tale-of-2-5-orthogonality-theses",0,"",""],["ELK shaving","Miss Aligned AI","2022","blog","LessWrong","www.lesswrong.com/posts/gZsTAsui5xqz7RTFt/elk-shaving",0,"","eliciting-latent-knowledge"],["Note-Taking without Hidden Messages","Hoagy","2022","blog","LessWrong","www.lesswrong.com/posts/czysbEDEsr9ijYeMd/note-taking-without-hidden-messages",0,"","eliciting-latent-knowledge"],["Quick Thoughts on A.I. Governance","NicholasKross","2022","blog","LessWrong","www.lesswrong.com/posts/mosYvGsKcpxvG4sTA/quick-thoughts-on-a-i-governance",0,"","governance"],["a unit for utils","Tamsin Leake","2022","blog","carado.moe","carado.moe/utils-unit.html",0,"",""],["Do FDT (or similar) recommend reparations?","David Scott Krueger (formerly: capybaralet)","2022","blog","LessWrong","www.lesswrong.com/posts/bEr9uGpxXHhsD6Adf/do-fdt-or-similar-recommend-reparations",0,"","theory"],["Learning the smooth prior","Geoffrey Irving and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9x5mtYjHYfr4T7KLj/learning-the-smooth-prior",0,"",""],["Prize for Alignment Research Tasks","stuhlmueller and William_S","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XLx3mpdi7HSp4rytF/prize-for-alignment-research-tasks",0,"","automated-alignment-research"],["the uncertainty of 2+2=4","Tamsin Leake","2022","blog","carado.moe","carado.moe/uncertainty-2+2=4.html",0,"",""],["Training Language Models with Language Feedback","Jérémy Scheurer","2022","paper","arXiv preprint","arxiv.org/abs/2204.14146",0,"","evals robustness"],["Slides: Potential Risks From Advanced AI","Aryeh Englander","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gAjdEcwg3DMSr4kLt/slides-potential-risks-from-advanced-ai",0,"",""],["[Intro to brain-like-AGI safety] 13. Symbol grounding & human social instincts","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/5F5Tz3u6kJbTNMqsb/intro-to-brain-like-agi-safety-13-symbol-grounding-and-human",0,"",""],["A Survey on XAI for Beyond 5G Security: Technical Aspects, Use Cases, Challenges and Research Directions","Thulitha Senevirathna and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.12822",0,"",""],["AI Alternative Futures: Scenario Mapping Artificial Intelligence Risk - Request for Participation (*Closed*)","Kakili","2022","blog","LessWrong","www.lesswrong.com/posts/7KfdM3wEeqJiwYcMN/ai-alternative-futures-scenario-mapping-artificial",0,"","instrumental-convergence governance forecasting"],["If you’re very optimistic about ELK then you should be optimistic about outer alignment","Sam Marks","2022","blog","LessWrong","www.lesswrong.com/posts/ZpQ3H4YgSw8BRZAHD/if-you-re-very-optimistic-about-elk-then-you-should-be",0,"","eliciting-latent-knowledge"],["Law-Following AI 1: Sequence Introduction and Structure","Cullen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NrtbF3JHFnqBCztXC/law-following-ai-1-sequence-introduction-and-structure",0,"","governance"],["Law-Following AI 2: Intent Alignment + Superintelligence → Lawless AI (By Default)","Cullen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9aSi7koXHCakb82Fz/law-following-ai-2-intent-alignment-superintelligence",0,"","governance"],["Law-Following AI 2: Intent Alignment + Superintelligence → Lawless AI (By Default)","Cullen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cEj7o9rbPjmy7CDht/law-following-ai-2-intent-alignment-superintelligence",0,"","governance policy"],["Law-Following AI 3: Lawless AI Agents Undermine Stabilizing Agreements","Cullen","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DfcXaGH7XGYjW22C2/law-following-ai-3-lawless-ai-agents-undermine-stabilizing",0,"","agents governance"],["Law-Following AI 3: Lawless AI Agents Undermine Stabilizing Agreements","Cullen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ExHkFcNAL9cjqFmsF/law-following-ai-3-lawless-ai-agents-undermine-stabilizing",0,"","agents governance policy"],["SERI ML Alignment Theory Scholars Program 2022","Ryan Kidd and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8vLvpxzpc6ntfBWNo/seri-ml-alignment-theory-scholars-program-2022",0,"",""],["SERI ML Alignment Theory Scholars Program 2022","Ryan Kidd and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nSyvMy3QQTyBzybNx/seri-ml-alignment-theory-scholars-program-2022",0,"",""],["The Speed + Simplicity Prior is probably anti-deceptive","anonymous","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/KSWSkxXJqWGd5jYLB/the-speed-simplicity-prior-is-probably-anti-deceptive",0,"","deception"],["[$20K in Prizes] AI Safety Arguments Competition","Dan H and 4 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3eP8D5Sxih3NhPE6F/usd20k-in-prizes-ai-safety-arguments-competition",0,"",""],["[$20K In Prizes] AI Safety Arguments Competition","ThomasW and 4 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/p3eiBqnijXPv5pCMA/usd20k-in-prizes-ai-safety-arguments-competition",0,"",""],["Framings of Deceptive Alignment","peterbarnett","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/8whGos5JCdBzDbZhH/framings-of-deceptive-alignment",0,"","alignment-faking deception"],["How to engage with AI 4 Social Justice actors","TomWestgarth","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cYveBTjXWoutARLvA/how-to-engage-with-ai-4-social-justice-actors",0,"",""],["Why Copilot Accelerates Timelines","Michaël Trazzi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/aqTAd7KzsYmHWYdei/why-copilot-accelerates-timelines",0,"","forecasting"],["[Request for Distillation] Coherence of Distributed Decisions With Different Inputs Implies Conditioning","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GnMWifHzAknqJsLnv/request-for-distillation-coherence-of-distributed-decisions",0,"",""],["Intuitions about solving hard problems","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GkXKvkLAcTm5ackCq/intuitions-about-solving-hard-problems",0,"",""],["Key questions about artificial sentience: an opinionated guide","rgb","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gFoWdiGYtXrhmBusH/key-questions-about-artificial-sentience-an-opinionated",0,"",""],["Make a neural network in ~10 minutes","Arjun Yadav","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MYuhZPSySQoo3h4kD/make-a-neural-network-in-10-minutes",0,"",""],["Towards Evaluating Adaptivity of Model-Based Reinforcement Learning Methods","Yi Wan and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.11464",0,"","evals deception"],["What is being improved in recursive self improvement?","Lone Pine","2022","blog","LessWrong","www.lesswrong.com/posts/bhBgjpZSAvxFGYn3s/what-is-being-improved-in-recursive-self-improvement",0,"","forecasting"],["Which Post Idea Is Most Effective?","Jordan Arel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/SLDcehczwEuBCt9T5/which-post-idea-is-most-effective",0,"",""],["Examining Evolution as an Upper Bound for AGI Timelines","meanderingmoose","2022","blog","LessWrong","www.lesswrong.com/posts/FmaeKTQgMpXfPDkfe/examining-evolution-as-an-upper-bound-for-agi-timelines",0,"","forecasting"],["Skilling-up in ML Engineering for Alignment: request for comments","TheMcDouglas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/jq5cbCxERw8t6PPQ8/skilling-up-in-ml-engineering-for-alignment-request-for",0,"",""],["[ASoT] Consequentialist models as a superset of mesaoptimizers","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Rhbac7CfRodMrs77F/asot-consequentialist-models-as-a-superset-of-mesaoptimizers",0,"",""],["Calling for Student Submissions: AI Safety Distillation Contest","Aris Richardson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ei4pYFJKcbGAdGnNb/calling-for-student-submissions-ai-safety-distillation",0,"",""],["Infra-Miscellanea","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SYeuMzspmwoQABWdw/infra-miscellanea",0,"",""],["Infra-Topology","Diffractor","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/PrYbdKcj89f8swCkr/infra-topology",0,"",""],["Instrumental Convergence To Offer Hope?","michael_mjd","2022","blog","LessWrong","www.lesswrong.com/posts/PfbE2nTvRJjtzysLM/instrumental-convergence-to-offer-hope",0,"","instrumental-convergence"],["Choice := Anthropics uncertainty? And potential implications for agency","Antoine de Scorraille","2022","blog","LessWrong","www.lesswrong.com/posts/JLNjNvsjyz6D9P2c7/choice-anthropics-uncertainty-and-potential-implications-for",0,"","agents theory"],["For every choice of AGI difficulty, conditioning on gradual take-off implies shorter timelines.","Francis Rhys Ward","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AiaAq5XeECg7MpTL7/for-every-choice-of-agi-difficulty-conditioning-on-gradual",0,"","forecasting"],["Path-Specific Objectives for Safer Agent Incentives","Sebastian Farquhar and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.10018",0,"","deception agents"],["The Risks of Machine Learning Systems","Samson Tan and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.09852",0,"",""],["What are the numbers in mind for the super-short AGI timelines so many long-termists are alarmed about?","Evan_Gaensbauer","2022","blog","LessWrong","www.lesswrong.com/posts/qDoqwGs4Dhj27sbTj/what-are-the-numbers-in-mind-for-the-super-short-agi",0,"","forecasting"],["[Intro to brain-like-AGI safety] 12. Two paths forward: “Controlled AGI” and “Social-instinct AGI”","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Sd4QvG4ZyjynZuHGt/intro-to-brain-like-agi-safety-12-two-paths-forward",0,"",""],["GPT-3 and concept extrapolation","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qyyo2efuwWkyR62fB/gpt-3-and-concept-extrapolation",0,"",""],["Why No *Interesting* Unaligned Singularity?","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/khupuW9cPrcLYkJww/why-no-interesting-unaligned-singularity",0,"",""],["[Closed] Hiring a mathematician to work on the learning-theoretic AI alignment agenda","Vanessa","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dcTnpX2AHXvYXg6wg/closed-hiring-a-mathematician-to-work-on-the-learning",0,"",""],["Another argument that you will let the AI out of the box","Garrett Baker","2022","blog","LessWrong","www.lesswrong.com/posts/7KmBfTjmRZSNaoCCi/another-argument-that-you-will-let-the-ai-out-of-the-box",0,"",""],["Chaining Retroactive Funders to Borrow Against Unlikely Utopias","Dawn Drescher","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6FdvKpxey9gLRe8S8/chaining-retroactive-funders-to-borrow-against-unlikely",0,"",""],["Concept extrapolation: key posts","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rPWQzRRQbjtgYn7rE/concept-extrapolation-key-posts",0,"",""],["Deceptive Agents are a Good Way to Do Things","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/MxXtaChisukWL5Ehz/deceptive-agents-are-a-good-way-to-do-things",0,"","deception agents robustness"],["“Pivotal Act” Intentions: Negative Consequences and Fallacious Arguments","Andrew_Critch","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Jo89KvfAs9z7owoZp/pivotal-act-intentions-negative-consequences-and-fallacious",0,"",""],["“Pivotal Act” Intentions: Negative Consequences and Fallacious Arguments","Andrew Critch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/q6t5zKCg5peZA92Zu/pivotal-act-intentions-negative-consequences-and-fallacious",0,"","governance"],["Hierarchical Optimal Transport for Comparing Histopathology Datasets","Anna Yeaton and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.08324",0,"",""],["How will the world respond to \"AI x-risk warning shots\" according to reference class forecasting?","Ryan Kidd","2022","blog","EA Forum","forum.effectivealtruism.org/posts/CyiuhttLjFuCygYoy/how-will-the-world-respond-to-ai-x-risk-warning-shots",0,"","forecasting"],["How I failed to form views on AI safety","Ada-Maaria Hyvärinen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ST3JjsLdTBnaK46BD/how-i-failed-to-form-views-on-ai-safety-3",0,"",""],["What is causality to an evidential decision theorist?","paulfchristiano","2022","blog","LessWrong","www.lesswrong.com/posts/2LisesnhDvRkqEMya/what-is-causality-to-an-evidential-decision-theorist",0,"","theory"],["Why not offer a multi-million / billion dollar prize for solving the Alignment Problem?","Aryeh Englander","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tSdEfPepkj6vHZyf9/why-not-offer-a-multi-million-billion-dollar-prize-for",0,"",""],["A grand strategy to recruit AI capabilities researchers into AI safety research","Peter S. Park","2022","blog","EA Forum","forum.effectivealtruism.org/posts/juhMehg89FrLX9pTj/a-grand-strategy-to-recruit-ai-capabilities-researchers-into",0,"",""],["Begging, Pleading AI Orgs to Comment on NIST AI Risk Management Framework","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/D8NfyuQeGspM9fYpT/begging-pleading-ai-orgs-to-comment-on-nist-ai-risk",0,"","evals policy"],["Contextualizing Artificially Intelligent Morality: A Meta-Ethnography of Top-Down, Bottom-Up, and Hybrid Models for Theoretical and Applied Ethics in Artificial Intelligence","Jennafer S. Roberts and Laura N. Montoya","2022","paper","arXiv preprint","arxiv.org/abs/2204.07612",0,"",""],["Everything I Need To Know About Takeoff Speeds I Learned From Air Conditioner Ratings On Amazon","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/MMAK6eeMCH3JGuqeZ/everything-i-need-to-know-about-takeoff-speeds-i-learned",0,"","forecasting"],["Please Share Your Perspectives on the Degree of Societal Impact from Transformative AI Outcomes","Kiliank","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nErbrZDTjzo8wxuvP/please-share-your-perspectives-on-the-degree-of-societal",0,"","forecasting"],["Refine: An Incubator for Conceptual Alignment Research Bets","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/D7epkkJb3CqDTYgX9/refine-an-incubator-for-conceptual-alignment-research-bets",0,"",""],["Some reasons why a predictor wants to be a consequentialist","Lauro Langosco","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DQBpu6LweoXyxLSsf/some-reasons-why-a-predictor-wants-to-be-a-consequentialist",0,"",""],["Early 2022 Paper Round-up","jsteinhardt","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qAhT2qvKXboXqLk4e/early-2022-paper-round-up",0,"",""],["How dath ilan coordinates around solving AI alignment","Thomas Kwa","2022","blog","EA Forum","forum.effectivealtruism.org/posts/EQEYFpGdf2evJmqCw/how-dath-ilan-coordinates-around-solving-ai-alignment",0,"",""],["Methodical Advice Collection and Reuse in Deep Reinforcement Learning","Sahir and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.07254",0,"","agents policy"],["Redwood Research is hiring for several roles (Operations and Technical)","JJXWang and billzito","2022","blog","EA Forum","forum.effectivealtruism.org/posts/M3N9ZXs8jmX8arYXK/redwood-research-is-hiring-for-several-roles-operations-and",0,"","robustness"],["Another list of theories of impact for interpretability","Beth Barnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YQALrtMkeqemAF5GX/another-list-of-theories-of-impact-for-interpretability",0,"","interpretability"],["Flexible Multiple-Objective Reinforcement Learning for Chip Placement","Fu-Chieh Chang and 13 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.06407",0,"",""],["Hierarchical text-conditional image generation with CLIP latents","OpenAI Research","2022","blog","openai.com","openai.com/research/hierarchical-text-conditional-image-generation-with-clip-latents",0,"",""],["Takeoff speeds have a huge effect on what it means to work on AI x-risk","Buck","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hRohhttbtpY3SHmmD/takeoff-speeds-have-a-huge-effect-on-what-it-means-to-work-1",0,"","forecasting"],["What more compute does for brain-like models: response to Rohin","Nathan Helm-Burger","2022","blog","LessWrong","www.lesswrong.com/posts/5Ae8rcYjWAe6zfdQs/what-more-compute-does-for-brain-like-models-response-to",0,"","forecasting"],["What to include in a guest lecture on existential risks from AI?","Aryeh Englander","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mwk8HGroo74WALLHy/what-to-include-in-a-guest-lecture-on-existential-risks-from",0,"",""],["“Fragility of Value” vs. LLMs","Not Relevant","2022","blog","LessWrong","www.lesswrong.com/posts/8Kxi3mEAwNxoYFu7T/fragility-of-value-vs-llms",0,"",""],["6 Year Decrease of Metaculus AGI Prediction","Chris Leong","2022","blog","EA Forum","forum.effectivealtruism.org/posts/nkwN4M6BcBmThoG7p/6-year-decrease-of-metaculus-agi-prediction",0,"","forecasting"],["A broad basin of attraction around human values?","Wei Dai","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TrvkWBwYvvJjSqSCj/a-broad-basin-of-attraction-around-human-values",0,"",""],["A primer & some reflections on recent CSER work (EAB talk)","MMMaas","2022","blog","EA Forum","forum.effectivealtruism.org/posts/WJZAc6fTYNbb5DeAW/a-primer-and-some-reflections-on-recent-cser-work-eab-talk",0,"","governance"],["A Small Negative Result on Debate","Sam Bowman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/CL8RFFCdsBwWWfKYS/a-small-negative-result-on-debate",0,"",""],["AdaTest:Reinforcement Learning and Adaptive Sampling for On-chip Hardware Trojan Detection","Huili Chen and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.06117",0,"","mechanistic-interpretability evals benchmarks"],["AI governance student hackathon on Saturday, April 23: register now!","mic","2022","blog","LessWrong","www.lesswrong.com/posts/N4wb4FbnuhaqSg7dT/ai-governance-student-hackathon-on-saturday-april-23",0,"","governance"],["An empirical analysis of compute-optimal large language model training","Jordan Hoffmann and 3 others","2022","blog","deepmind.com","www.deepmind.com/blog/an-empirical-analysis-of-compute-optimal-large-language-model-training",0,"",""],["finding earth in the universal program","Tamsin Leake","2022","blog","carado.moe","carado.moe/finding-earth-ud.html",0,"",""],["Help us find pain points in AI safety","Esben Kran","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LqXk92pK5xeph4eDB/help-us-find-pain-points-in-ai-safety",0,"",""],["How to become an AI safety researcher","peterbarnett","2022","blog","EA Forum","forum.effectivealtruism.org/posts/PH2pqsqgXQkfCdmkv/how-to-become-an-ai-safety-researcher",0,"",""],["Is technical AI alignment research a net positive?","cranberry_bear","2022","blog","LessWrong","www.lesswrong.com/posts/eaCcc7AhQj4EoHFzH/is-technical-ai-alignment-research-a-net-positive",0,"",""],["Jaime Sevilla - Projecting AI progress from compute┬átrends-by Towards Data Science-video_id 2NXagVA3yzg-date 20220413","Jaime Sevilla and Jeremie Harris","2022","report","drive.google.com","drive.google.com/file/d/15lAvcmvMDJ34gCQuxcTR-kyV5Y-NC2C3/view?usp=share_link",0,"",""],["Ought's theory of change","stuhlmueller and jungofthewon","2022","blog","EA Forum","forum.effectivealtruism.org/posts/raFAKyw7ofSo9mRQ3/ought-s-theory-of-change",0,"",""],["Reward model hacking as a challenge for reward learning","Erik Jenner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yyoKYmfFx7zpPyD99/reward-model-hacking-as-a-challenge-for-reward-learning",0,"",""],["Three questions about mesa-optimizers","Eric Neyman","2022","blog","LessWrong","www.lesswrong.com/posts/PrNw3EBSwYfJyREjE/three-questions-about-mesa-optimizers",0,"",""],["Tips for conducting worldview investigations","lukeprog","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vcjLwqLDqNEmvewHY/tips-for-conducting-worldview-investigations",0,"","governance"],["Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","Yuntao Bai and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.05862",0,"","rlhf evals policy robustness"],["Useful Vices for Wicked Problems","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/useful-vices-for-wicked-problems/",0,"",""],["An AI-in-a-box success model","azsantosk","2022","blog","LessWrong","www.lesswrong.com/posts/ubCjGESWpKJrTBXTP/an-ai-in-a-box-success-model",0,"",""],["Credo AI is hiring for several roles","IanEisenberg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/MDtbDMNvaJsb75FiD/credo-ai-is-hiring-for-several-roles",0,"","governance"],["Goodhart's Law Causal Diagrams","JustinShovelain and Jeremy Gillen","2022","blog","LessWrong","www.lesswrong.com/posts/e4SMfYWb4Tz568yh6/goodhart-s-law-causal-diagrams",0,"","goodharts-law"],["Linguistic communication as (inverse) reward design","Theodore R. Sumers and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.05091",0,"","agents"],["Metaethical Perspectives on 'Benchmarking' AI Ethics","Travis LaCroix and Alexandra Sasha Luccioni","2022","paper","arXiv preprint","arxiv.org/abs/2204.05151",0,"","evals benchmarks"],["The Peerless","Tamsin Leake","2022","blog","carado.moe","carado.moe/the-peerless.html",0,"",""],["The Regulatory Option: A response to near 0% survival odds","Matthew Lowenstein","2022","blog","LessWrong","www.lesswrong.com/posts/9oxKJghFjk9WmG9YX/the-regulatory-option-a-response-to-near-0-survival-odds",0,"","governance"],["A visualization of some orgs in the AI Safety Pipeline","Aaron_Scher and Aman Patel","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DWSTgzjApuEjMdsyY/a-visualization-of-some-orgs-in-the-ai-safety-pipeline",0,"",""],["Crucial considerations in the field of Wild Animal Welfare (WAW)","Holly_Elmore","2022","blog","EA Forum","forum.effectivealtruism.org/posts/z63MFmYXSCHeFxRz3/crucial-considerations-in-the-field-of-wild-animal-welfare",0,"",""],["Enhancing the Robustness, Efficiency, and Diversity of Differentiable Architecture Search","Chao Li and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.04681",0,"","evals robustness"],["A concrete bet offer to those with short AGI timelines","Matthew Barnett and Tamay","2022","blog","LessWrong","www.lesswrong.com/posts/X3p8mxE5dHYDZNxCm/a-concrete-bet-offer-to-those-with-short-agi-timelines",0,"","forecasting"],["A tough career decision","PabloAMC","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pPrFJdRq7aPu8pFo3/a-tough-career-decision",0,"",""],["AMA Conjecture, A New Alignment Startup","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rtEtTybuCcDWLk7N9/ama-conjecture-a-new-alignment-startup",0,"",""],["bracing for the alignment tunnel","Tamsin Leake","2022","blog","carado.moe","carado.moe/bracing-alignment-tunnel.html",0,"",""],["Elicit: Language Models as Research Assistants","stuhlmueller and jungofthewon","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/s5jrfbsGLyEexh4GT/elicit-language-models-as-research-assistants",0,"",""],["Hyperbolic takeoff","Ege Erdil","2022","blog","LessWrong","www.lesswrong.com/posts/etgJYbkvvkBoDRm4k/hyperbolic-takeoff",0,"","forecasting"],["The right to protection from catastrophic AI risk","Jack Cunningham","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QA4N75QwsCbZtFBWF/the-right-to-protection-from-catastrophic-ai-risk",0,"",""],["[RETRACTED] It's time for EA leadership to pull the short-timelines fire alarm.","Not Relevant","2022","blog","LessWrong","www.lesswrong.com/posts/wrkEnGrTTrM2mnmGa/retracted-it-s-time-for-ea-leadership-to-pull-the-short",0,"","forecasting"],["AIs should learn human preferences, not biases","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3PTfkdLLqZE9vppXC/ais-should-learn-human-preferences-not-biases",0,"",""],["Different perspectives on concept extrapolation","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/j9vCEjRFDwmH8FTKH/different-perspectives-on-concept-extrapolation",0,"",""],["Emergent Ventures AI","Gavin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/qdkow9kQhuqtoyxxs/emergent-ventures-ai",0,"",""],["Language Model Tools for Alignment Research","Logan Riggs","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/AhF8iXLu5PchsmyKf/language-model-tools-for-alignment-research",0,"",""],["We Are Conjecture, A New Alignment Research Startup","Connor Leahy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jfq2BH5kfQqu2vYv3/we-are-conjecture-a-new-alignment-research-startup",0,"",""],["[ASoT] Some thoughts about imperfect world modeling","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/bzLpXZMGAiMdfLNKy/asot-some-thoughts-about-imperfect-world-modeling",0,"",""],["How BoMAI Might fail","Donald Hobson","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FebeDdcToayY6rHSf/how-bomai-might-fail",0,"",""],["Ideal governance (for companies, countries and more)","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hxTFAetiiSL7dZmyb/ideal-governance-for-companies-countries-and-more",0,"","governance policy theory"],["Is GPT3 a Good Rationalist? - InstructGPT3 [2/2]","simeon_c","2022","blog","LessWrong","www.lesswrong.com/posts/a3FuA7fGgpTQ7mX3W/is-gpt3-a-good-rationalist-instructgpt3-2-2",0,"","interpretability eliciting-latent-knowledge robustness"],["New Sequence - Towards a worldwide, watertight Windfall Clause","John Bridge","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DJuhFbtJLJ92pCsKW/new-sequence-towards-a-worldwide-watertight-windfall-clause",0,"","governance policy"],["Productive Mistakes, Not Perfect Answers","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ADMWDDKGgivgghxWf/productive-mistakes-not-perfect-answers",0,"",""],["Robust Event-Driven Interactions in Cooperative Multi-Agent Learning","Daniel Jarne Ornia and Manuel Mazo Jr","2022","paper","Formal Modeling and Analysis of Timed Systems. FORMATS 2022.\n  Lecture Notes in Computer Science, vol 13465. Springer, Cham","arxiv.org/abs/2204.03361",0,"","deception agents robustness"],["Truthfulness, standards and credibility","Joe_Collman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Brr84ZmvK3kwy2eGJ/truthfulness-standards-and-credibility",0,"",""],["What Should We Optimize - A Conversation","Johannes C. Mayer","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zCBZb5M2wTzngayNc/what-should-we-optimize-a-conversation",0,"",""],["[Cross-post] Change my mind: we should define and measure the effectiveness of advanced AI","David Johnston","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vxK7zuhfhYuoohGR7/cross-post-change-my-mind-we-should-define-and-measure-the",0,"",""],["[Intro to brain-like-AGI safety] 11. Safety ≠ alignment (but they’re close!)","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BeQcPCTAikQihhiaK/intro-to-brain-like-agi-safety-11-safety-alignment-but-they",0,"",""],["[Link] A minimal viable product for alignment","janleike","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fYf9JAwa6BYMt8GBj/link-a-minimal-viable-product-for-alignment",0,"","automated-alignment-research"],["[Link] Why I’m excited about AI-assisted human feedback","janleike","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qunrsimS2cECxyCKy/link-why-i-m-excited-about-ai-assisted-human-feedback",0,"","rlhf"],["A Cognitive Framework for Delegation Between Error-Prone AI and Human Agents","Andrew Fuchs and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.02889",0,"","agents"],["PaLM in \"Extrapolating GPT-N performance\"","Lukas Finnveden","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YzbQeCiwoLBHrvAh4/palm-in-extrapolating-gpt-n-performance",0,"",""],["Testing PaLM prompts on GPT3","Yitz","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/EHbJ69JDs4suovpLw/testing-palm-prompts-on-gpt3",0,"",""],["AXRP Episode 14 - Infra-Bayesian Physicalism with Vanessa Kosoy","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YZ7xzwDrqrvnZjwdn/axrp-episode-14-infra-bayesian-physicalism-with-vanessa",0,"",""],["Ideal governance (for companies, countries and more)","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/ideal-governance-for-companies-countries-and-more/",0,"","governance"],["Supervise Process, not Outcomes","stuhlmueller and jungofthewon","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pYcFPMBtQveAjcSfH/supervise-process-not-outcomes",0,"",""],["The case for Doing Something Else (if Alignment is doomed)","Rafael Harth","2022","blog","LessWrong","www.lesswrong.com/posts/yhRTjBs6oiNcjRgcx/the-case-for-doing-something-else-if-alignment-is-doomed",0,"","governance"],["What an actually pessimistic containment strategy looks like","lc","2022","blog","LessWrong","www.lesswrong.com/posts/kipMvuaK3NALvFHc9/what-an-actually-pessimistic-containment-strategy-looks-like",0,"","governance"],["Yudkowsky and Christiano on AI Takeoff Speeds [LINKPOST]","aogara","2022","blog","EA Forum","forum.effectivealtruism.org/posts/x3MSTGhxChEZsRvwB/yudkowsky-and-christiano-on-ai-takeoff-speeds-linkpost",0,"","forecasting"],["Call For Distillers","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zo9zKcz47JxDErFzQ/call-for-distillers",0,"",""],["Project Intro: Selection Theorems for Modularity","TheMcDouglas and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XKwKJCXgSKhSr9bZY/project-intro-selection-theorems-for-modularity",0,"",""],["Theories of Modularity in the Biological Literature","TheMcDouglas and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JzTfKrgC7Lfz3zcwM/theories-of-modularity-in-the-biological-literature",0,"",""],["AI Governance across Slow/Fast Takeoff and Easy/Hard Alignment spectra","Davidmanheim","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xxMYFKLqiBJZRNoPj/ai-governance-across-slow-fast-takeoff-and-easy-hard",0,"","governance forecasting"],["Is it valuable to the field of AI Safety to have a neuroscience background?","Samuel Nellessen","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3izcm9sBmTPbRtQNH/is-it-valuable-to-the-field-of-ai-safety-to-have-a",0,"",""],["On Agent Incentives to Manipulate Human Feedback in Multi-Agent Reward Learning Scenarios","Francis Rhys Ward","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/6TxmJRDGzDbwcLE3w/on-agent-incentives-to-manipulate-human-feedback-in-multi",0,"","rlhf agents"],["Optimality is the tiger, and agents are its teeth","Veedrac","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kpPnReyBC54KESiSn/optimality-is-the-tiger-and-agents-are-its-teeth",0,"","agents"],["What are the best ideas of how to regulate AI from the US executive branch?","Jack Cunningham","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j72evEkLAuAmndkTC/what-are-the-best-ideas-of-how-to-regulate-ai-from-the-us",0,"","policy"],["CHAI Newsletter #1 2022","CHAI","2022","report","drive.google.com","drive.google.com/file/d/1sb1IlXM1FMU6lYEYLSZODama2EuHSdK2/view?usp=sharing",0,"",""],["Graph-in-Graph (GiG): Learning interpretable latent graphs in non-Euclidean domain for biological and healthcare applications","Kamilia Mullakaeva and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2204.00323",0,"","interpretability"],["Jeff Shainline thinks that there is too much serendipity in the physics of optical/superconducting computing, suggesting that they were part of the criteria of Cosmological Natural Selection, which could have some fairly lovecraftian implications","mako yass","2022","blog","LessWrong","www.lesswrong.com/posts/EEoxY5YyqpTMcjJoz/jeff-shainline-thinks-that-there-is-too-much-serendipity-in",0,"","forecasting"],["New Scaling Laws for Large Language Models","1a3orn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/midXmMb2Xg37F2Kgn/new-scaling-laws-for-large-language-models",0,"","scaling-laws"],["Questions about ''formalizing instrumental goals\"","Mark Neyer","2022","blog","LessWrong","www.lesswrong.com/posts/ELvmLtY8Zzcko9uGJ/questions-about-formalizing-instrumental-goals",0,"","instrumental-convergence"],["Replacing Karma with Good Heart Tokens (Worth $1!)","Ben Pace and habryka","2022","blog","LessWrong","www.lesswrong.com/posts/mz3hwS4c9bc9EHAm9/replacing-karma-with-good-heart-tokens-worth-usd1",0,"","goodharts-law robustness"],["[Link] Training Compute-Optimal Large Language Models","nostalgebraist","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/4dbK5dPiqHCgNdKnq/link-training-compute-optimal-large-language-models",0,"","forecasting scaling-laws"],["AXRP Episode 13 - First Principles of AGI Safety with Richard Ngo","DanielFilan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tEf8fEFCkFtPyg9pm/axrp-episode-13-first-principles-of-agi-safety-with-richard",0,"",""],["[ASoT] Some thoughts about LM monologue limitations and ELK","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Jiy3n5KMsGGJ6NNYH/asot-some-thoughts-about-lm-monologue-limitations-and-elk",0,"","eliciting-latent-knowledge"],["[Intro to brain-like-AGI safety] 10. The alignment problem","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wucncPjud27mLWZzQ/intro-to-brain-like-agi-safety-10-the-alignment-problem",0,"","reward-hacking goodharts-law instrumental-convergence"],["Announcing the EU Tech Policy Fellowship","Jan-Willem and 2 others","2022","blog","EA Forum","forum.effectivealtruism.org/posts/iTcdun6jm9nxLy8Rp/announcing-the-eu-tech-policy-fellowship",0,"","governance policy robustness"],["ELK Computational Complexity: Three Levels of Difficulty","abramdemski","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ztnkDKD5odorWt5dB/elk-computational-complexity-three-levels-of-difficulty",0,"","eliciting-latent-knowledge"],["Is AI safety still neglected?","Coafos","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HYYJnAtmoavcbksgp/is-ai-safety-still-neglected",0,"",""],["No, EDT Did Not Get It Right All Along: Why the Coin Flip Creation Problem Is Irrelevant","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/hEiAPeirmbKB4FeWW/no-edt-did-not-get-it-right-all-along-why-the-coin-flip",0,"","theory"],["Pitching AI Safety in 3 sentences","PabloAMC","2022","blog","EA Forum","forum.effectivealtruism.org/posts/vCAvL3DLhf2peuEhR/pitching-ai-safety-in-3-sentences",0,"",""],["Procedurally evaluating factual accuracy: a request for research","Jacob_Hilton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2zEeb36XL6HLnjDkj/procedurally-evaluating-factual-accuracy-a-request-for",0,"","evals"],["Request for Assistance - Research on Scenario Development for Advanced AI Risk","Kiliank","2022","blog","EA Forum","forum.effectivealtruism.org/posts/coJu2iSwCDf2z5LJH/request-for-assistance-research-on-scenario-development-for",0,"",""],["8 possible high-level goals for work on nuclear risk","MichaelA","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dASEFCurRpNot4Gpc/8-possible-high-level-goals-for-work-on-nuclear-risk",0,"",""],["A dataset for AI/superintelligence stories and other media?","Harrison Durland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/McMwdgLZnuiCdsrrb/a-dataset-for-ai-superintelligence-stories-and-other-media",0,"",""],["An Artificial Intelligence Browser Architecture (AIBA) For Our Kind and Others: A Voice Name System Speech implementation with two warrants, Wake Neutrality and Value Preservation of Personally Identifiable Information","Brian Subirana","2022","paper","arXiv preprint","arxiv.org/abs/2203.16497",0,"","training-data"],["Can we simulate human evolution to create a somewhat aligned AGI?","Thomas Kwa","2022","blog","EA Forum","forum.effectivealtruism.org/posts/uk39s7wgLcotmFsto/can-we-simulate-human-evolution-to-create-a-somewhat-aligned",0,"",""],["Debating myself on whether “extra lives lived” are as good as “deaths prevented”","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/debating-myself-on-whether-extra-lives-lived-are-as-good-as-deaths-prevented/",0,"","robustness"],["Gears-Level Mental Models of Transformer Interpretability","KevinRoWang","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/X26ksz4p3wSyycKNB/gears-level-mental-models-of-transformer-interpretability",0,"","interpretability"],["Quality Assurance of Generative Dialog Models in an Evolving Conversational Agent Used for Swedish Language Practice","Markus Borg and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.15414",0,"","agents assurance"],["should we implement free will?","Tamsin Leake","2022","blog","carado.moe","carado.moe/implement-free-will.html",0,"",""],["Towards a better circuit prior: Improving on ELK state-of-the-art","evhub and kcwoolverton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/7ygmXXGjXZaEktF6M/towards-a-better-circuit-prior-improving-on-elk-state-of-the",0,"","mechanistic-interpretability eliciting-latent-knowledge"],["What would make you confident that AGI has been achieved?","Yitz","2022","blog","LessWrong","www.lesswrong.com/posts/bpJ3A5sDoBq6i83Xp/what-would-make-you-confident-that-agi-has-been-achieved",0,"","forecasting"],["[ASoT] Some thoughts about deceptive mesaoptimization","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oijJc8Mu2jPNgpuvy/asot-some-thoughts-about-deceptive-mesaoptimization",0,"","deception"],["A Primer on God, Liberalism and the End of History","Mahdi Complex","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AzwXdHqAkBhbHSQZL/a-primer-on-god-liberalism-and-the-end-of-history",0,"","governance"],["AI safety starter pack","mariushobbhahn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/pbiGHk6AjRxdBPoD8/ai-safety-starter-pack",0,"",""],["Seeking Survey Responses - Attitudes Towards AI risks","anson","2022","blog","EA Forum","forum.effectivealtruism.org/posts/7JeqhfQq9QTtkEbjL/seeking-survey-responses-attitudes-towards-ai-risks",0,"",""],["The role of academia in AI Safety.","PabloAMC","2022","blog","EA Forum","forum.effectivealtruism.org/posts/g2nFCW5xYqgtrWfJ9/the-role-of-academia-in-ai-safety",0,"","red-teaming"],["Vaniver's ELK Submission","Vaniver","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/9f4zBjiFndqbR8y6e/vaniver-s-elk-submission",0,"","eliciting-latent-knowledge"],["[ASoT] Searching for consequentialist structure","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/f6ByNdGJYxR3Kwguy/asot-searching-for-consequentialist-structure",0,"",""],["[ASoT] Some ways ELK could still be solvable in practice","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/SbxWdhhwJWCpifTst/asot-some-ways-elk-could-still-be-solvable-in-practice",0,"","eliciting-latent-knowledge"],["Practical everyday human strategizing","anonymous","2022","blog","LessWrong","www.lesswrong.com/posts/3dwADq2hjsJB2GAno/practical-everyday-human-strategizing",0,"","goodharts-law"],["Scenario Mapping Advanced AI Risk: Request for Participation with Data Collection","Kiliank","2022","blog","EA Forum","forum.effectivealtruism.org/posts/m8PsJsSfQAYxPusHi/scenario-mapping-advanced-ai-risk-request-for-participation",0,"","governance"],["[ASoT] Observations about ELK","leogao","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kWTko53s2DqTeprjz/asot-observations-about-elk",0,"","eliciting-latent-knowledge"],["Compute Governance: The Role of Commodity Hardware","Jan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/z8BF9GwcCjeXShC4q/compute-governance-the-role-of-commodity-hardware",0,"","governance compute-governance"],["When people ask for your P(doom), do you give them your inside view or your betting odds?","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/sAEE7fdnv3KpcaQEi/when-people-ask-for-your-p-doom-do-you-give-them-your-inside",0,"",""],["I'm interviewing Nova Das Sarma about AI safety and information security. What shouId I ask her?","Robert_Wiblin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LM6TxGTJhhyixD5of/i-m-interviewing-nova-das-sarma-about-ai-safety-and",0,"",""],["What's the best machine learning newsletter? How do you keep up to date?","Mathieu Putz","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dotnvSqB2faF3kHcs/what-s-the-best-machine-learning-newsletter-how-do-you-keep",0,"",""],["Why Agent Foundations? An Overly Abstract Explanation","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FWvzwCDRgcjb9sigb/why-agent-foundations-an-overly-abstract-explanation",0,"","goodharts-law agents theory"],["A Rationale-Centric Framework for Human-in-the-loop Machine Learning","Jinghui Lu and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.12918",0,"","benchmarks deception robustness"],["AI Safety Overview: CERI Summer Research Fellowship","Jamie Bernardi","2022","blog","EA Forum","forum.effectivealtruism.org/posts/T3piiDHvaGuzE7KKF/ai-safety-overview-ceri-summer-research-fellowship-1",0,"","governance"],["Data Publication for the 2021 Artificial Intelligence, Morality, and Sentience (AIMS) Survey","Janet Pauketat","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BajkGx4FbaE9y3vz5/data-publication-for-the-2021-artificial-intelligence",0,"","governance"],["Interview with bj9ne","bj9ne and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1v72aCvHRgQpAzFvt468iDM2GA6qOvyuD/view",0,"",""],["Interview with cvgig","cvgig and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/15KPTTyONZkBjGyeO1J7NN_dptUGXwY9V/view",0,"",""],["On expected utility, part 4: Dutch books, Cox, and Complete Class","Joe Carlsmith","2022","blog","LessWrong","www.lesswrong.com/posts/wf3BqEWrwbQj3fksF/on-expected-utility-part-4-dutch-books-cox-and-complete",0,"","theory"],["Your Policy Regulariser is Secretly an Adversary","DeepMind Safety Research","2022","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/your-policy-regulariser-is-secretly-an-adversary-14684c743d45",0,"","policy"],["[Intro to brain-like-AGI safety] 9. Takeaways from neuro 2/2: On AGI motivation","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vpdJz4k5BgGzuGo7A/intro-to-brain-like-agi-safety-9-takeaways-from-neuro-2-2-on",0,"","reward-hacking interpretability"],["A survey of tool use and workflows in alignment research","Logan Riggs and 3 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ebYiodG3MAEqskCDG/a-survey-of-tool-use-and-workflows-in-alignment-research-1",0,"","tool-use automated-alignment-research"],["Meditations on careers in AI Safety","PabloAMC","2022","blog","EA Forum","forum.effectivealtruism.org/posts/THyzvDPThjK2P8fn3/meditations-on-careers-in-ai-safety",0,"",""],["NeurIPSorICML_bj9ne-by Vael Gates-date 20220324","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1pZqoqMcr_gKlkiLwLMsn9D9MwDsTzHHX/view?usp=share_link",0,"",""],["NeurIPSorICML_cvgig-by Vael Gates-date 20220324","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1gFuZ9_ykohGQTTCnbCnReg5saTR9Letl/view?usp=share_link",0,"",""],["goals for emergency unaligned AI","Tamsin Leake","2022","blog","carado.moe","carado.moe/emergency-unaligned-ai-goals.html",0,"",""],["Interview with lgu5f","lgu5f and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1vhh9YN7ybAX9-hyYCZimEogitAGaC7RE/view",0,"",""],["are there finitely many moral patients?","Tamsin Leake","2022","blog","carado.moe","carado.moe/finite-patients.html",0,"",""],["Desirable? AI qualities","brb243","2022","blog","EA Forum","forum.effectivealtruism.org/posts/BPcwvcqvuzScrfhrf/desirable-ai-qualities",0,"",""],["E.A. Megaproject Ideas","Tomer_Goloboy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Khz5s6hrTWo4cReNL/e-a-megaproject-ideas",0,"","policy"],["Interview with 92iem","92iem and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1-1YpMOCJJKAXdQDVlhbNf68f9_Bh_ADU/view",0,"",""],["making the UD and UDASSA less broken: identifying time steps","Tamsin Leake","2022","blog","carado.moe","carado.moe/udassa-time-steps.html",0,"",""],["NeurIPSorICML_lgu5f-by Vael Gates-date 20220322","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1OPwFpaoRgP0P05ciMIobJUh_mq2aRhNb/view?usp=share_link",0,"",""],["the word \"syntax\" in programming, linguistics and LISP","Tamsin Leake","2022","blog","carado.moe","carado.moe/the-word-syntax.html",0,"",""],["values system as test-driven development","Tamsin Leake","2022","blog","carado.moe","carado.moe/values-tdd.html",0,"",""],["ViM: Out-Of-Distribution with Virtual-logit Matching","Haoqi Wang","2022","paper","arXiv preprint","arxiv.org/abs/2203.10807",0,"","evals benchmarks robustness"],["How might a herd of interns help with AI or biosecurity research tasks/questions?","Harrison Durland","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HZacQkvLLeLKT3a6j/how-might-a-herd-of-interns-help-with-ai-or-biosecurity",0,"",""],["individuallyselected_92iem-by Vael Gates-date 20220321","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1yZ9Kn2-yBMhi9r4a4XoAo2zBRdJuYs29/view?usp=share_link",0,"",""],["Interview with 7oalk","7oalk and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1rZxNoVYm2-sOEc6FQXIT3c9Qq0mgdN8p/view",0,"",""],["Natural Value Learning","Chris van Merwijk","2022","blog","LessWrong","www.lesswrong.com/posts/WoXDjh3Bqnene9uxA/natural-value-learning",0,"",""],["Robust Action Gap Increasing with Clipped Advantage Learning","Zhe Zhang and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.11677",0,"","benchmarks robustness"],["What EAG sessions would you like on AI?","Nathan Young","2022","blog","EA Forum","forum.effectivealtruism.org/posts/6Qw3JvEDkAzmaqpTK/what-eag-sessions-would-you-like-on-ai",0,"",""],["Can you be Not Even Wrong in AI Alignment?","throwaway8238","2022","blog","LessWrong","www.lesswrong.com/posts/DbfQrno5W6SDT9ABc/can-you-be-not-even-wrong-in-ai-alignment",0,"","eliciting-latent-knowledge"],["Exploring Finite Factored Sets with some toy examples","Thomas Kehrenberg","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hxuKtHH4jTdtmEAbK/exploring-finite-factored-sets-with-some-toy-examples",0,"",""],["NeurIPSorICML_7oalk-by Vael Gates-date 20220320","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1WIx_cUCJ-eCVQ_YgLNTZGsc5cgMqjBGY/view?usp=share_link",0,"",""],["Wargaming AGI Development","ryan_b","2022","blog","LessWrong","www.lesswrong.com/posts/ouFnZoYaKqicC6jH8/wargaming-agi-development",0,"","forecasting"],["Career Advice: Philosophy + Programming -> AI Safety","tcelferact","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hBjrAeuGvwk9pbwLL/career-advice-philosophy-programming-greater-than-ai-safety-1",0,"",""],["Cold Takes reader survey - let me know what you want more and less of!","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/cold-takes-reader-survey-let-me-know-what-you-want-more-and-less-of/",0,"",""],["Interview with 7ujun","7ujun and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/17rhNp735EyyI7R0bdMXVIEw1QkpWM1Yi/view",0,"",""],["Interview with 84py7","84py7 and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1khOhU_4oVuHRblLnDQxtzntCBBmVlIsV/view",0,"",""],["Interview with a0nfw","a0nfw and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/18vny49YcNuDePGyXY27mz0j0cwI0lw9G/view",0,"",""],["Interview with q243b","q243b and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1fXWGvs_Vsp8CihQk-73wfv-xzm6xsDEW/view",0,"",""],["Interview with w5cb5","w5cb5 and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1GN_ZHi5Jx7NEvPrL8gT0I9em_av1ZyVk/view",0,"",""],["Interview with zlzai","zlzai and Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1vM40bOBHmMkXaJwVvIJCJMRg76EKxlhF/view",0,"",""],["AI Risk Management Framework: Initial Draft","NIST","2022","report","nist.gov","www.nist.gov/system/files/documents/2022/03/17/AI-RMF-1stdraft.pdf",0,"",""],["individuallyselected_7ujun-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1GViBUPA6EYawSuVc67rE6TsTojvrZIRO/view?usp=share_link",0,"",""],["individuallyselected_84py7-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1l2mL8og3xsto-XlUF_r1AYgAtqvbceJ7/view?usp=share_link",0,"",""],["individuallyselected_w5cb5-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1gONSbSzr8dA2BJlFjxqXGt6BlyUKEpxR/view?usp=share_link",0,"",""],["individuallyselected_zlzai-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/11aBW7_2Y6WyY9MDmaLwC1Ark-VDUu0EV/view?usp=share_link",0,"",""],["NeurIPSorICML_a0nfw-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1YxuzV0J6k1naLHt_GSRubthL-Rh9mxsu/view?usp=share_link",0,"",""],["NeurIPSorICML_q243b-by Vael Gates-date 20220318","Vael Gates","2022","report","drive.google.com","drive.google.com/file/d/1ldx2yW-B9KzpTk6tNlejikFfdKVyCDQz/view?usp=share_link",0,"",""],["[Intro to brain-like-AGI safety] 8. Takeaways from neuro 1/2: On AGI development","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fDPsYdDtkzhBp9A8D/intro-to-brain-like-agi-safety-8-takeaways-from-neuro-1-2-on",0,"",""],["Building AI Innovation Labs together with Companies","Jens Heidrich and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.08465",0,"","evals"],["Danger(s) of theorem-proving AI?","Yitz","2022","blog","LessWrong","www.lesswrong.com/posts/Xb6RGvTzbcHhJ4jXR/danger-s-of-theorem-proving-ai",0,"",""],["GopherCite: Teaching language models to support answers with verified quotes","Jacob Menick and 10 others","2022","blog","deepmind.com","www.deepmind.com/blog/gophercite-teaching-language-models-to-support-answers-with-verified-quotes",0,"",""],["Mediocre AI safety as existential risk","Gavin","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j4G5Gqxa6JmbbQYzX/mediocre-ai-safety-as-existential-risk",0,"",""],["Resilient Neural Forecasting Systems","Michael Bohlke-Schneider and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.08492",0,"","forecasting robustness"],["Teaching language models to support answers with verified quotes","Jacob Menick and 10 others","2022","report","storage.googleapis.com","storage.googleapis.com/deepmind-media/Teaching%20language%20models%20to%20support%20answers%20with%20verified%20quotes/Teaching%20language%20models%20to%20support%20answers%20with%20verified%20quotes.pdf",0,"",""],["Dual use of artificial-intelligence-powered drug discovery","Vaniver","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/YQhBhxFhChGExS5HE/dual-use-of-artificial-intelligence-powered-drug-discovery",0,"",""],["Early-warning Forecasting Center: What it is, and why it'd be cool","Linch","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zjMeGcgWpvDcm3CkH/early-warning-forecasting-center-what-it-is-and-why-it-d-be",0,"","forecasting"],["ELK contest submission: route understanding through the human ontology","Vika and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QrhCsuaEmSLzc8NQ4/elk-contest-submission-route-understanding-through-the-human",0,"","eliciting-latent-knowledge"],["There should be an AI safety project board","mariushobbhahn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/srzs5smvt5FvhfFS5/there-should-be-an-ai-safety-project-board",0,"",""],["Twitter-length responses to 24 AI alignment arguments","RobBensinger","2022","blog","EA Forum","forum.effectivealtruism.org/posts/FzcQSpbiiom7RHEjD/twitter-length-responses-to-24-ai-alignment-arguments",0,"",""],["Algebraic Learning: Towards Interpretable Information Modeling","Tong Owen Yang","2022","paper","arXiv preprint","arxiv.org/abs/2203.06690",0,"","interpretability deception"],["CMKD: CNN/Transformer-Based Cross-Model Knowledge Distillation for Audio Classification","Yuan Gong and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.06760",0,"","robustness"],["Label-only Model Inversion Attack: The Attack that Requires the Least Information","Dayong Ye and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.06555",0,"",""],["New GPT3 Impressive Capabilities - InstructGPT3 [1/2]","simeon_c","2022","blog","EA Forum","forum.effectivealtruism.org/posts/o7ouaa7Bbx6dKJdQC/new-gpt3-impressive-capabilities-instructgpt3-1-2",0,"",""],["Compute Trends — Comparison to OpenAI’s AI and Compute","lennart and 5 others","2022","blog","LessWrong","www.lesswrong.com/posts/sDiGGhpw7Evw7zdR4/compute-trends-comparison-to-openai-s-ai-and-compute",0,"","scaling-laws"],["Followup on Terminator","skluug","2022","blog","EA Forum","forum.effectivealtruism.org/posts/dr2ig3tquB59viY2v/followup-on-terminator",0,"",""],["A Longlist of Theories of Impact for Interpretability","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/uK6sQCNMw8WKzJeCQ/a-longlist-of-theories-of-impact-for-interpretability",0,"","interpretability"],["[Intro to brain-like-AGI safety] 7. From hardcoded drives to foresighted plans: A worked example","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zXibERtEWpKuG5XAC/intro-to-brain-like-agi-safety-7-from-hardcoded-drives-to",0,"",""],["A Rephrasing Of and Footnote To An Embedded Agency Proposal","JoshuaOSHickman","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mTuQDeuiXKnk972WR/a-rephrasing-of-and-footnote-to-an-embedded-agency-proposal",0,"","theory"],["Ask AI companies about what they are doing for AI safety?","mic","2022","blog","LessWrong","www.lesswrong.com/posts/7dfqwqJWEbP6p8Qzx/ask-ai-companies-about-what-they-are-doing-for-ai-safety",0,"",""],["ELK prize results","paulfchristiano and Mark Xu","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zjMKpSB2Xccn9qi5t/elk-prize-results",0,"","eliciting-latent-knowledge"],["ELK Sub - Note-taking in internal rollouts","Hoagy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/fftQP7zrnYkDqgwfj/elk-sub-note-taking-in-internal-rollouts",0,"","eliciting-latent-knowledge"],["It Looks Like You're Trying To Take Over The World","gwern","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/a5e9arCnbDac9Doig/it-looks-like-you-re-trying-to-take-over-the-world",0,"","forecasting"],["On presenting the case for AI risk","Aryeh Englander","2022","blog","LessWrong","www.lesswrong.com/posts/jcA4rath4HvFtmdm6/on-presenting-the-case-for-ai-risk",0,"",""],["Programming note","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/programming-note/",0,"",""],["Towards a Roadmap on Software Engineering for Responsible AI","Qinghua Lu and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.08594",0,"","governance"],["“Intro to brain-like-AGI safety” series—halfway point!","Steven Byrnes","2022","blog","EA Forum","forum.effectivealtruism.org/posts/8Ajsy96jGHcf27Xre/intro-to-brain-like-agi-safety-series-halfway-point",0,"","forecasting"],["[MLSN #3]: NeurIPS Safety Paper Roundup","Dan H","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dhbLE8BqRvhBtsXhS/mlsn-3-neurips-safety-paper-roundup",0,"",""],["AI Risk is like Terminator; Stop Saying it's Not","skluug","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zsFCj2mfnYZmSW2FF/ai-risk-is-like-terminator-stop-saying-it-s-not-1",0,"","instrumental-convergence"],["In-context Learning and Induction Heads","Catherine Olsson and 25 others","2022","blog","transformer-circuits.pub","transformer-circuits.pub/2022/in-context-learning-and-induction-heads/index.html",0,"",""],["Irina Rish - Out-of-distribution generalization-by Towards Data Science-video_id QjXFN4UWZCg-date 20220309","Irina Rish and Jeremie Harris","2022","report","drive.google.com","drive.google.com/file/d/12EKK8SfY21Tge1M9ADJHYUk2yWB7T4zG/view?usp=share_link",0,"",""],["ML Safety Newsletter #3","Dan Hendrycks","2022","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-3",0,"",""],["Value extrapolation, concept extrapolation, model splintering","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/i8sHdLyGQeBTGwTqq/value-extrapolation-concept-extrapolation-model-splintering",0,"",""],["An Intuitive Introduction to Causal Decision Theory","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/bNayfvnKKsbE7w6Sb/an-intuitive-introduction-to-causal-decision-theory",0,"","theory"],["An Intuitive Introduction to Evidential Decision Theory","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/pQfAmKtQTf8ndB9cw/an-intuitive-introduction-to-evidential-decision-theory",0,"","theory"],["An Intuitive Introduction to Functional Decision Theory","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/d5swTmH2zw4vzYBNS/an-intuitive-introduction-to-functional-decision-theory",0,"","theory"],["Basic Concepts in Decision Theory","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/nojovDKpf9fRAzhwy/basic-concepts-in-decision-theory",0,"","theory"],["Projecting compute trends in Machine Learning","Tamay and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3dBtgKCkJh5yCHbag/projecting-compute-trends-in-machine-learning-2",0,"","forecasting"],["Enabling Automated Machine Learning for Model-Driven AI Engineering","Armin Moin and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.02927",0,"",""],["experience/moral patient deduplication and ethics","Tamsin Leake","2022","blog","carado.moe","carado.moe/deduplication-ethics.html",0,"",""],["Preserving and continuing alignment research through a severe global catastrophe","A_donor","2022","blog","LessWrong","www.lesswrong.com/posts/xrxh3usuoYMckkKom/preserving-and-continuing-alignment-research-through-a",0,"",""],["Why work at AI Impacts?","Katja Grace","2022","blog","aiimpacts.org","aiimpacts.org/why-work-at-ai-impacts/",0,"",""],["Is transformative AI the biggest existential risk? Why or why not?","BrownHairedEevee","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DspD7yA87zwhcSAt3/is-transformative-ai-the-biggest-existential-risk-why-or-why",0,"",""],["A Typology for Exploring the Mitigation of Shortcut Behavior","Felix Friedrich and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.03668",0,"","evals benchmarks"],["AutoDIME: Automatic Design of Interesting Multi-Agent Environments","Ingmar Kanitscheider and Harri Edwards","2022","paper","arXiv preprint","arxiv.org/abs/2203.02481",0,"","evals agents policy robustness"],["recognition","Tamsin Leake","2022","blog","carado.moe","carado.moe/recognition.html",0,"",""],["A research agenda for assessing the economic impacts of code generation models","Gillian Hadfield and 2 others","2022","blog","openai.com","openai.com/research/economic-impacts",0,"",""],["Credo AI is hiring!","IanEisenberg","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JmzTk4GBzRQj4NeLb/credo-ai-is-hiring",0,"","governance"],["Graph Neural Networks for Multimodal Single-Cell Data Integration","Hongzhi Wen and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.01884",0,"",""],["Learning Robust Real-Time Cultural Transmission without Human Data","Cultural General Intelligence Team","2022","blog","deepmind.com","www.deepmind.com/blog/learning-robust-real-time-cultural-transmission-without-human-data",0,"",""],["Reasoning about Counterfactuals to Improve Human Inverse Reinforcement Learning","Michael S. Lee and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.01855",0,"","agents"],["What will be some of the most impactful applications of advanced AI in the near term?","IanDavidMoss","2022","blog","EA Forum","forum.effectivealtruism.org/posts/5hKHqoqvGDgZiuoHL/what-will-be-some-of-the-most-impactful-applications-of",0,"","forecasting"],["3D Common Corruptions and Data Augmentation","Oğuzhan Fatih Kar and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.01441",0,"","evals benchmarks robustness"],["[Intro to brain-like-AGI safety] 6. Big picture of motivation, decision-making, and RL","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qNZSBqLEh4qLRqgWW/intro-to-brain-like-agi-safety-6-big-picture-of-motivation",0,"",""],["do not hold on to your believed intrinsic values — follow your heart!","Tamsin Leake","2022","blog","carado.moe","carado.moe/not-hold-on-to-values.html",0,"",""],["Musings on the Speed Prior","evhub","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/GC69Hmc6ZQDM9xC3w/musings-on-the-speed-prior",0,"","eliciting-latent-knowledge"],["Ngo and Yudkowsky on scientific reasoning and pivotal acts","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/03/01/ngo-and-yudkowsky-on-scientific-reasoning-and-pivotal-acts/",0,"",""],["Ordinary and unordinary decision theory","JonasMoss","2022","blog","LessWrong","www.lesswrong.com/posts/uS6vdQH8zpHyMAsxR/ordinary-and-unordinary-decision-theory",0,"","theory"],["Responsible-AI-by-Design: a Pattern Collection for Designing Responsible AI Systems","Qinghua Lu and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2203.00905",0,"","governance"],["Shah and Yudkowsky on alignment failures","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/03/02/shah-and-yudkowsky-on-alignment-failures/",0,"",""],["The Wicked Problem Experience","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/the-wicked-problem-experience/",0,"",""],["Would (myopic) general public good producers significantly accelerate the development of AGI?","mako yass","2022","blog","LessWrong","www.lesswrong.com/posts/PGfJPnDzy9sDE6zkj/would-myopic-general-public-good-producers-significantly",0,"","forecasting robustness"],["[Link] Aligned AI AMA","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZgBRkddr9ckxCyxji/link-aligned-ai-ama",0,"",""],["AGI x-risk timelines: 10% chance (by year X) estimates should be the headline, not 50%.","Greg_Colbourn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/9BPs6ZmtqCbNfYaKg/agi-x-risk-timelines-10-chance-by-year-x-estimates-should-be",0,"","forecasting"],["AI Ethical Committee","eaaicommittee","2022","blog","EA Forum","forum.effectivealtruism.org/posts/mno4DMuHEWxLCQKXv/ai-ethical-committee",0,"","governance"],["AI Value Alignment Speaker Series Presented By EA Berkeley","Mahendra Prasad","2022","blog","EA Forum","forum.effectivealtruism.org/posts/HatYvQkGFMCj2BnzH/ai-value-alignment-speaker-series-presented-by-ea-berkeley",0,"",""],["AI views and disagreements AMA: Christiano, Ngo, Shah, Soares, Yudkowsky","RobBensinger","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tCmPDx7bFDYkmpAY7/ai-views-and-disagreements-ama-christiano-ngo-shah-soares",0,"",""],["Christiano and Yudkowsky on AI predictions and human intelligence","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/03/01/christiano-and-yudkowsky-on-ai-predictions-and-human-intelligence/",0,"",""],["February 2022 Newsletter","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/03/01/february-2022-newsletter/",0,"",""],["Being an individual alignment grantmaker","A_donor","2022","blog","EA Forum","forum.effectivealtruism.org/posts/eDwcke3TbbZKAYkgi/being-an-individual-alignment-grantmaker",0,"",""],["ELK Thought Dump","abramdemski","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eqzbXmqGqXiyjX3TP/elk-thought-dump-1",0,"","eliciting-latent-knowledge"],["Late 2021 MIRI Conversations: AMA / Discussion","Rob Bensinger","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/34Gkqus9vusXRevR8/late-2021-miri-conversations-ama-discussion",0,"",""],["Shah and Yudkowsky on alignment failures","Rohin Shah and Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/tcCxPLBrEXdxN5HCQ/shah-and-yudkowsky-on-alignment-failures",0,"",""],["Shah and Yudkowsky on alignment failures","EliezerYudkowsky and Rohin Shah","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DuPEzGJ5oscqxD5oh/shah-and-yudkowsky-on-alignment-failures",0,"",""],["The dangers in algorithms learning humans' values and irrationalities","Rebecca Gorman and Stuart Armstrong","2022","paper","arXiv preprint","arxiv.org/abs/2202.13985",0,"","policy"],["How I Formed My Own Views About AI Safety","Neel Nanda","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JZrN4ckaCfd6J37cG/how-i-formed-my-own-views-about-ai-safety",0,"",""],["How do new models from OpenAI, DeepMind and Anthropic perform on TruthfulQA?","Owain_Evans","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/yYkrbS5iAwdEQyynW/how-do-new-models-from-openai-deepmind-and-anthropic-perform",0,"",""],["IMO challenge bet with Eliezer","paulfchristiano","2022","blog","LessWrong","www.lesswrong.com/posts/sWLLdG6DWJEy3CH7n/imo-challenge-bet-with-eliezer",0,"","forecasting"],["New Speaker Series on AI Alignment Starting March 3","Zechen Zhang","2022","blog","EA Forum","forum.effectivealtruism.org/posts/QYHJ6GSkusS7EjbSg/new-speaker-series-on-ai-alignment-starting-march-3",0,"",""],["The Quest for a Common Model of the Intelligent Decision Maker","Richard S. Sutton","2022","paper","arXiv preprint","arxiv.org/abs/2202.13252",0,"","evals agents"],["University community building seems like the wrong model for AI safety","George Stiffman","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YBD9BoDaapCfqBmNd/university-community-building-seems-like-the-wrong-model-for",0,"",""],["Composing Complex and Hybrid AI Solutions","Peter Schüller and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.12566",0,"",""],["OCR-IDL: OCR Annotations for Industry Document Library Dataset","Ali Furkan Biten and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.12985",0,"","training-data"],["Re: Some thoughts on vegetarianism and veganism","Fai","2022","blog","EA Forum","forum.effectivealtruism.org/posts/AZyJdher64htcpKti/re-some-thoughts-on-vegetarianism-and-veganism",0,"",""],["The Big Picture Of Alignment (Talk Part 2)","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/aEtc5GgqJGFtTH2kQ/the-big-picture-of-alignment-talk-part-2-1",0,"",""],["The “Slicing Problem” for Computational Theories of Consciousness","Andrés Gómez-Emilsson and Chris Percy","2022","report","degruyter.com","www.degruyter.com/document/doi/10.1515/opphil-2022-0225/html",0,"",""],["Trust-maximizing AGI","Jan and Karl von Wendt","2022","blog","LessWrong","www.lesswrong.com/posts/6xiBgLvvDiH7Sboq2/trust-maximizing-agi",0,"",""],["A comment on Ajeya Cotra's draft report on AI timelines","Matthew Barnett","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qnjDGitKxYaesbsem/a-comment-on-ajeya-cotra-s-draft-report-on-ai-timelines",0,"","forecasting"],["All You Need Is Supervised Learning: From Imitation Learning to Meta-RL With Upside Down RL","Kai Arulkumaran and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.11960",0,"","agents policy"],["Important, actionable research questions for the most important century","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zGiD94SHwQ9MwPyfW/important-actionable-research-questions-for-the-most",0,"","governance"],["Transformer inductive biases & RASP","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kwpvEpDXsivbmdYhr/transformer-inductive-biases-and-rasp",0,"",""],["[Intro to brain-like-AGI safety] 5. The “long-term predictor”, and TD learning","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/F759WQ8iKjqBncDki/intro-to-brain-like-agi-safety-5-the-long-term-predictor-and",0,"",""],["Christiano and Yudkowsky on AI predictions and human intelligence","Eliezer Yudkowsky","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/NbGmfxbaABPsspib7/christiano-and-yudkowsky-on-ai-predictions-and-human",0,"",""],["Christiano and Yudkowsky on AI predictions and human intelligence","EliezerYudkowsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/RNgbY3zCS4CGSqKGm/christiano-and-yudkowsky-on-ai-predictions-and-human",0,"",""],["Drawing Inductor Layout with a Reinforcement Learning Agent: Method and Application for VCO Inductors","Cameron Haigh and 7 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.11798",0,"","mechanistic-interpretability agents"],["More GPT-3 and symbol grounding","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QppXf4yfcG8JAKhnw/more-gpt-3-and-symbol-grounding",0,"",""],["my current pyramid of needs","Tamsin Leake","2022","blog","carado.moe","carado.moe/pyramid-needs.html",0,"",""],["Probing Image-Language Transformers for Verb Understanding","Lisa Anne Hendricks and Aida Nematzadeh","2022","blog","deepmind.com","www.deepmind.com/blog/probing-image-language-transformers-for-verb-understanding",0,"","interpretability"],["ELK Proposal: Thinking Via A Human Imitator","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/z3xTDPDsndJBmHLFH/elk-proposal-thinking-via-a-human-imitator",0,"","eliciting-latent-knowledge"],["Learning By Writing","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/learning-by-writing/",0,"",""],["Retrieval Augmented Classification for Long-Tail Visual Recognition","Alexander Long and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.11233",0,"","evals training-data"],["Alignment research exercises","Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kj37Hzb2MsALwLqWt/alignment-research-exercises",0,"",""],["Favorite / most obscure research on understanding DNNs?","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/269kZdpdKtmLwHni4/favorite-most-obscure-research-on-understanding-dnns",0,"",""],["HCMD-zero: Learning Value Aligned Mechanisms from Data","Jan Balaguer and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.10122",0,"","interpretability agents policy"],["Inferring Lexicographically-Ordered Rewards from Preferences","Alihan Hüyük and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.10153",0,"","agents"],["Investigations of Performance and Bias in Human-AI Teamwork in Hiring","Andi Peng and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.11812",0,"",""],["Ngo and Yudkowsky on scientific reasoning and pivotal acts","Eliezer Yudkowsky and Richard_Ngo","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/cCrpbZ4qTCEYXbzje/ngo-and-yudkowsky-on-scientific-reasoning-and-pivotal-acts",0,"",""],["Ngo and Yudkowsky on scientific reasoning and pivotal acts","EliezerYudkowsky and richard_ngo","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fH266hKDhJMFKBSgs/ngo-and-yudkowsky-on-scientific-reasoning-and-pivotal-acts-1",0,"",""],["The Big Picture Of Alignment (Talk Part 1)","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/xdSDFQs4aC5GrdHNZ/the-big-picture-of-alignment-talk-part-1",0,"",""],["Two Challenges for ELK","derek shiller","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rxQbX2JpigjnbnL3A/two-challenges-for-elk",0,"","eliciting-latent-knowledge"],["Deconstructing Distributions: A Pointwise Framework of Learning","Gal Kaplun and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.09931",0,"","evals robustness"],["Alignment researchers, how useful is extra compute for you?","Lauro Langosco","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/A4djH6sc9vZq2AYBD/alignment-researchers-how-useful-is-extra-compute-for-you-1",0,"",""],["Analogy of AI Alignment as Raising a Child?","Aaron_Scher","2022","blog","EA Forum","forum.effectivealtruism.org/posts/zH88C83bnPtLruwKg/analogy-of-ai-alignment-as-raising-a-child",0,"",""],["HCH and Adversarial Questions","David Udell","2022","blog","LessWrong","www.lesswrong.com/posts/picPfLnygZC5aFjNr/hch-and-adversarial-questions",0,"","scalable-oversight"],["Thoughts on Dangerous Learned Optimization","peterbarnett","2022","blog","LessWrong","www.lesswrong.com/posts/rzJ9FgCoxuqSR2zb5/thoughts-on-dangerous-learned-optimization",0,"",""],["Critical Checkpoints for Evaluating Defence Models Against Adversarial Attack and Robustness","Kanak Tekwani and Manojkumar Parmar","2022","paper","arXiv preprint","arxiv.org/abs/2202.09039",0,"","evals robustness"],["Implications of automated ontology identification","Alex Flint and 2 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/LRgM9cuLNPbsjwEdN/implications-of-automated-ontology-identification",0,"","eliciting-latent-knowledge"],["Misc thematic links","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/misc-thematic-links/",0,"",""],["REPL's and ELK","scottviteri","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/C5PZNi5fueH2RC6aF/repl-s-and-elk",0,"","eliciting-latent-knowledge"],["[Intro to brain-like-AGI safety] 4. The “short-term predictor”","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y3bkJ59j4dciiLYyw/intro-to-brain-like-agi-safety-4-the-short-term-predictor",0,"",""],["[Intro to brain-like-AGI safety] 4. The “short-term predictor”","Steven Byrnes","2022","blog","LessWrong","www.lesswrong.com/posts/Y3bkJ59j4dciiLYyw/intro-to-brain-like-agi-safety-4-the-short-term-predictor",0,"",""],["Compute Trends Across Three eras of Machine Learning","Jsevillamol and 5 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XKtybmbjhC6mXDm5z/compute-trends-across-three-eras-of-machine-learning",0,"","forecasting scaling-laws"],["Defending One-Dimensional Ethics","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/defending-one-dimensional-ethics/",0,"",""],["How harmful are improvements in AI? + Poll","tilmanr and Marius Hobbhahn","2022","blog","LessWrong","www.lesswrong.com/posts/uABbabv5WPZmwzCmP/how-harmful-are-improvements-in-ai-poll",0,"","governance forecasting"],["Is ELK enough? Diamond, Matrix and Child AI","adamShimi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/XjDcwtgkHGWYA7stn/is-elk-enough-diamond-matrix-and-child-ai",0,"","eliciting-latent-knowledge"],["Predictability and Surprise in Large Generative Models","Deep Ganguli and 29 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.07785",0,"","policy scaling-laws"],["REPL's: a type signature for agents","scottviteri","2022","blog","LessWrong","www.lesswrong.com/posts/kN2cFPaLQhExEzgeZ/repl-s-a-type-signature-for-agents",0,"","eliciting-latent-knowledge agents"],["Safe Reinforcement Learning by Imagining the Near Future","Garrett Thomas and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.07789",0,"","deception agents"],["Some Hacky ELK Ideas","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3gAKoaziTXmvHusRv/some-hacky-elk-ideas",0,"","eliciting-latent-knowledge"],["What Does The Natural Abstraction Framework Say About ELK?","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HuqwRug3v6z3gEgKK/what-does-the-natural-abstraction-framework-say-about-elk",0,"","eliciting-latent-knowledge"],["Zero-Shot Assistance in Sequential Decision Problems","Sebastiaan De Peuter and Samuel Kaski","2022","paper","arXiv preprint","arxiv.org/abs/2202.07364",0,"","agents"],["[Linkpost] How To Get Into Independent Research On Alignment/Agency","Jackson Wagner","2022","blog","EA Forum","forum.effectivealtruism.org/posts/x3ih5ohtTdLXQf4Fq/linkpost-how-to-get-into-independent-research-on-alignment",0,"",""],["A Map to Navigate AI Governance","CaroJ","2022","blog","EA Forum","forum.effectivealtruism.org/posts/tmxkRFx6HyhhvHdz4/a-map-to-navigate-ai-governance",0,"","evals governance policy"],["Question 5: The timeline hyperparameter","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/BSNFKi3aym7DtSnTX/question-5-the-timeline-hyperparameter",0,"","forecasting"],["A Simplified Variant of Gödel's Ontological Argument","Christoph Benzmüller","2022","paper","arXiv preprint","arxiv.org/abs/2202.06264",0,"",""],["Abstractions as Redundant Information","johnswentworth","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/vvEebH5jEvxnJEvBC/abstractions-as-redundant-information",0,"",""],["Is a career in making AI systems more secure a meaningful way to mitigate the X-risk posed by AGI?","Kyle O’Brien","2022","blog","EA Forum","forum.effectivealtruism.org/posts/hBAeEJqunNKXv8Mnp/is-a-career-in-making-ai-systems-more-secure-a-meaningful",0,"",""],["Question 4: Implementing the control proposals","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/RgWFCDntyc3DEfgLn/question-4-implementing-the-control-proposals",0,"","governance"],["Defending against Adversarial Policies in Reinforcement Learning with Alternating Training","sergia","2022","blog","EA Forum","forum.effectivealtruism.org/posts/YscrJFofd6S8eJGS8/defending-against-adversarial-policies-in-reinforcement",0,"",""],["Question 3: Control proposals for minimizing bad outcomes","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/QgsH9yWBvFtuZDsgN/question-3-control-proposals-for-minimizing-bad-outcomes",0,"","interpretability"],["Uncalibrated Models Can Improve Human-AI Collaboration","Kailas Vodrahalli and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.05983",0,"",""],["Online Decision Transformer","Qinqing Zheng and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.05607",0,"","benchmarks"],["Predicting Out-of-Distribution Error with the Projection Norm","Yaodong Yu and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.05834",0,"","robustness"],["Question 2: Predicted bad outcomes of AGI learning architecture","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/e9MbFLBAnGkEfPTde/question-2-predicted-bad-outcomes-of-agi-learning",0,"",""],["To Match the Greats, Don’t Follow In Their Footsteps","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/to-match-the-greats-dont-follow-in-their-footsteps/",0,"",""],["A summary of aligning narrowly superhuman models","gugu","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/TSxAXeHHhgSxR5wGZ/a-summary-of-aligning-narrowly-superhuman-models",0,"",""],["Inferring utility functions from locally non-transitive preferences","Jan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QZiGEDiobFz8ropA5/inferring-utility-functions-from-locally-non-transitive",0,"",""],["Interpretable pipelines with evolutionarily optimized modules for RL tasks with visual inputs","Leonardo Lucio Custode and Giovanni Iacca","2022","paper","arXiv preprint","arxiv.org/abs/2202.04943",0,"","interpretability benchmarks"],["Locating and Editing Factual Associations in GPT","Kevin Meng and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.05262",0,"","evals"],["Proceedings of the Robust Artificial Intelligence System Assurance (RAISA) Workshop 2022","","2022","paper","arXiv preprint","arxiv.org/abs/2202.04787",0,"","evals assurance robustness"],["Question 1: Predicted architecture of AGI learning algorithm(s)","Cameron Berg","2022","blog","LessWrong","www.lesswrong.com/posts/snwpyAfzoFKdfnEDj/question-1-predicted-architecture-of-agi-learning-algorithm",0,"",""],["Trust in AI: Interpretability is not necessary or sufficient, while black-box interaction is necessary and sufficient","Max W. Shen","2022","paper","arXiv preprint","arxiv.org/abs/2202.05302",0,"","interpretability evals"],["[Intro to brain-like-AGI safety] 3. Two subsystems: Learning & Steering","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hE56gYi5d68uux9oM/intro-to-brain-like-agi-safety-3-two-subsystems-learning-and",0,"","forecasting"],["An extension of Aumann's approach for reducing game theory to bayesian decision theory to include EDT and UDT like agents","Karl Brisebois","2022","blog","LessWrong","www.lesswrong.com/posts/qEFG8BK9HCnKcHuGH/an-extension-of-aumann-s-approach-for-reducing-game-theory",0,"","agents theory"],["\"Moral progress\" vs. the simple passage of time","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/moral-progress-vs-the-simple-passage-of-time/",0,"",""],["Defending Functional Decision Theory","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/Suk3qEWyxnTG47TDZ/defending-functional-decision-theory",0,"","theory"],["How complex are myopic imitators?","Vivek Hebbar","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/2eRgFFeeS7pR4R8nD/how-complex-are-myopic-imitators-1",0,"",""],["Hypothesis: gradient descent prefers general circuits","Quintin Pope","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/JFibrXBewkSDmixuo/hypothesis-gradient-descent-prefers-general-circuits",0,"","mechanistic-interpretability"],["Local Explanations for Reinforcement Learning","Ronny Luss and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.03597",0,"","deception policy robustness"],["Machine Explanations and Human Understanding","Chacha Chen and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.04092",0,"",""],["Metaculus launches contest for essays with quantitative predictions about AI","Tamay Besiroglu and Metaculus","2022","blog","LessWrong","www.lesswrong.com/posts/j5shgF5LJC75GoXrt/metaculus-launches-contest-for-essays-with-quantitative",0,"","forecasting"],["Paradigm-building: Introduction","Cameron Berg","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/4TuzWEKysvYdhRXLd/paradigm-building-introduction",0,"",""],["Software engineering - Career review","Benjamin Hilton and 80000_Hours","2022","blog","EA Forum","forum.effectivealtruism.org/posts/gbPthwLw3NovHAJdp/software-engineering-career-review",0,"",""],["Red Teaming Language Models with Language Models","Ethan Perez and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.03286",0,"","evals red-teaming training-data"],["Red Teaming Language Models with Language Models","Ethan Perez and 8 others","2022","blog","deepmind.com","www.deepmind.com/blog/red-teaming-language-models-with-language-models",0,"","red-teaming"],["forking bitrate and entropy control","Tamsin Leake","2022","blog","carado.moe","carado.moe/forking-bitrate-entropy-control.html",0,"",""],["Human rights, democracy, and the rule of law assurance framework for AI systems: A proposal","David Leslie and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.02776",0,"","interpretability assurance governance"],["Science Facing Interoperability as a Necessary Condition of Success and Evil","Remy Demichelis","2022","paper","arXiv preprint","arxiv.org/abs/2202.02540",0,"",""],["Alignment versus AI Alignment","Alex Flint","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hTfyX4823wKqnoFnS/alignment-versus-ai-alignment",0,"",""],["Anti-Parfit's Hitchhiker","k64","2022","blog","LessWrong","www.lesswrong.com/posts/8mHtoM5gaW2QsL82c/anti-parfit-s-hitchhiker",0,"","theory"],["balancing utilitarianism","Tamsin Leake","2022","blog","carado.moe","carado.moe/balancing-utilitarianism.html",0,"",""],["Do mesa-optimization problems correlate with low-slack?","sudo -i","2022","blog","LessWrong","www.lesswrong.com/posts/jmrTMNhA4sKcrGEzu/do-mesa-optimization-problems-correlate-with-low-slack",0,"",""],["Knowledge-Integrated Informed AI for National Security","Anu K. Myne and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.03188",0,"",""],["political technology","Tamsin Leake","2022","blog","carado.moe","carado.moe/political-technology.html",0,"",""],["The 6-Ds of Creating AI-Enabled Systems","John Piorkowski","2022","paper","arXiv preprint","arxiv.org/abs/2202.03172",0,"",""],["Certifying Out-of-Domain Generalization for Blackbox Functions","Maurice Weber and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.01679",0,"","assurance robustness"],["hackable multiverse","Tamsin Leake","2022","blog","carado.moe","carado.moe/hackable-multiverse.html",0,"",""],["Investigating musical genius by listening to the Beach Boys a lot","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/investigating-musical-genius-by-listening-to-the-beach-boys-a-lot/",0,"",""],["Observed patterns around major technological advancements","richardkorzekwa","2022","blog","aiimpacts.org","aiimpacts.org/observed-patterns-around-major-technological-advancements/",0,"",""],["QNR prospects are important for AI alignment research","Eric Drexler","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/FKE6cAzQxEK4QH9fC/qnr-prospects-are-important-for-ai-alignment-research",0,"","interpretability"],["Reward is not enough: can we liberate AI from the reinforcement learning paradigm?","Vacslav Glukhov","2022","paper","arXiv preprint","arxiv.org/abs/2202.03192",0,"","agents"],["Technology Ethics in Action: Critical and Interdisciplinary Perspectives","Ben Green","2022","paper","Special Issue of the Journal of Social Computing (2021)","arxiv.org/abs/2202.01351",0,"","governance"],["The Met Dataset: Instance-level Recognition for Artworks","Nikolaos-Antonios Ypsilantis and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.01747",0,"","evals benchmarks robustness"],["[Intro to brain-like-AGI safety] 2. “Learning from scratch” in the brain","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/wBHSYwqssBGCnwvHg/intro-to-brain-like-agi-safety-2-learning-from-scratch-in",0,"",""],["a cognitively hazardous idea","Tamsin Leake","2022","blog","carado.moe","carado.moe/a-cognitively-hazardous-idea.html",0,"",""],["Announcing GPT-NeoX-20B","Connor Leahy","2022","blog","blog.eleuther.ai","blog.eleuther.ai/announcing-20b/",0,"",""],["Future-proof ethics","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/future-proof-ethics/",0,"",""],["Impossibility results for unbounded utilities","paulfchristiano","2022","blog","LessWrong","www.lesswrong.com/posts/hbmsW2k9DxED5Z4eJ/impossibility-results-for-unbounded-utilities",0,"","theory"],["OpenAI Solves (Some) Formal Math Olympiad Problems","Michaël Trazzi","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/q3vAgFnbDja9hZm9E/openai-solves-some-formal-math-olympiad-problems",0,"",""],["Thoughts on AGI safety from the top","jylin04","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/ApLnWjgMwBTJt6buC/thoughts-on-agi-safety-from-the-top",0,"","forecasting"],["VOS: Learning What You Don't Know by Virtual Outlier Synthesis","Xuefeng Du and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.01197",0,"","robustness"],["CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","Michael Laskin and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.00161",0,"","evals benchmarks"],["Interactive configurator with FO(.) and IDP-Z3","Pierre Carbonnelle and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.00343",0,"",""],["January 2022 Newsletter","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2022/01/31/january-2022-newsletter/",0,"",""],["AMA: Future of Life Institute's EU Team","Risto Uuk","2022","blog","EA Forum","forum.effectivealtruism.org/posts/j5xhPbj7ywdv6aEJc/ama-future-of-life-institute-s-eu-team",0,"","governance policy"],["Argument Against Impact: EU Is Not an AI Superpower","EU AI Governance","2022","blog","EA Forum","forum.effectivealtruism.org/posts/suyb4vC75Wo9EKgyu/argument-against-impact-eu-is-not-an-ai-superpower",0,"","governance forecasting"],["Should you work in the European Union to do AGI governance?","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/fbG6wWZhJ3jt3xHxS/should-you-work-in-the-european-union-to-do-agi-governance",0,"","governance policy"],["Explaining Reinforcement Learning Policies through Counterfactual Trajectories","Julius Frost and 6 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.12462",0,"","interpretability agents policy robustness"],["Certifying Model Accuracy under Distribution Shifts","Aounon Kumar and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.12440",0,"","robustness"],["Towards Safe Reinforcement Learning with a Safety Editor Policy","Haonan Yu and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.12427",0,"","agents policy"],["Arguments about Highly Reliable Agent Designs as a Useful Path to Artificial Intelligence Safety","riceissa and Davidmanheim","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/hWtpqjYXAvFExmAsD/arguments-about-highly-reliable-agent-designs-as-a-useful",0,"","agents theory"],["Causality, Transformative AI and alignment - part I","Marius Hobbhahn","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/oqzasmQ9Lye45QDMZ/causality-transformative-ai-and-alignment-part-i",0,"",""],["Cost disease and civilizational decline","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/cost-disease-and-civilizational-decline/",0,"",""],["Human-centered mechanism design with Democratic AI","Raphael Koster and 10 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.11441",0,"","policy"],["Newcomb's Lottery Problem","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/yk5iRtFKesLe6i6sE/newcomb-s-lottery-problem",0,"","theory"],["[Intro to brain-like-AGI safety] 1. What's the problem & Why work on it now?","Steven Byrnes","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/4basF9w9jaPZpoC8R/intro-to-brain-like-agi-safety-1-what-s-the-problem-and-why",0,"",""],["Cybertrust: From Explainable to Actionable and Interpretable AI (AI2)","Stephanie Galaitsi and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.11117",0,"","interpretability"],["ELK First Round Contest Winners","Mark Xu and paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/qXFbGzS3Sg2NhrNAu/elk-first-round-contest-winners",0,"","eliciting-latent-knowledge"],["Mo Gawdat - Scary Smart - A former Google exec_s perspective on AI┬árisk-by Towards Data Science-video_id u2cK0_jUX_g-date 20220126","Mo Gawdat and Jeremie Harris","2022","report","drive.google.com","drive.google.com/file/d/1icPC_qIAlhQ_75_-IrTGUKs7du4njv35/view?usp=share_link",0,"",""],["Reader reactions and update on \"Where's Today's Beethoven\"","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/reader-reactions-and-update-on-wheres-todays-beethoven/",0,"",""],["Safe AI -- How is this Possible?","Harald Rueß and Simon Burton","2022","paper","arXiv preprint","arxiv.org/abs/2201.10436",0,"","deception"],["CSER is hiring for a senior research associate on longterm AI risk and governance","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/z9JdTZnFDwf7FBMCc/cser-is-hiring-for-a-senior-research-associate-on-longterm",0,"","governance"],["Scaling Up Knowledge Graph Creation to Large and Heterogeneous Data Sources","Enrique Iglesias and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.09694",0,"","evals benchmarks deception"],["Alignment Problems All the Way Down","peterbarnett","2022","blog","LessWrong","www.lesswrong.com/posts/2CFBi4MNFNQXdbkss/alignment-problems-all-the-way-down",0,"",""],["Instrumental Convergence For Realistic Agent Objectives","TurnTrout","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/W22Btd7NmGuucFejc/instrumental-convergence-for-realistic-agent-objectives",0,"","instrumental-convergence agents"],["[AN #171]: Disagreements between alignment \"optimists\" and \"pessimists\"","Rohin Shah","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/3vFmQhHBosnjZXuAJ/an-171-disagreements-between-alignment-optimists-and",0,"",""],["Avoiding Unsafe States in 3D Environments using Human Feedback","DeepMind Safety Research","2022","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/avoiding-unsafe-states-in-3d-environments-using-human-feedback-5869ed9fb94c",0,"","rlhf"],["Identifying Adversarial Attacks on Text Classifiers","Zhouhang Xie and 8 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.08555",0,"","benchmarks"],["Sharing Powerful AI Models","Alexis Carlier","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/czRtKPj3qC3wi5i94/sharing-powerful-ai-models",0,"",""],["[linkpost] Sharing powerful AI models: the emerging paradigm of structured access","ts","2022","blog","EA Forum","forum.effectivealtruism.org/posts/i8Eseu6HXHKp37Hye/linkpost-sharing-powerful-ai-models-the-emerging-paradigm-of",0,"",""],["Action: Help expand funding for AI Safety by coordinating on NSF response","Evan R. Murphy","2022","blog","EA Forum","forum.effectivealtruism.org/posts/NzuJmTtRfJBjxcmnD/action-help-expand-funding-for-ai-safety-by-coordinating-on",0,"",""],["Book non-review: The Dawn of Everything","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/book-non-review-the-dawn-of-everything/",0,"",""],["Estimating training compute of Deep Learning models","lennart and 4 others","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/HvqQm6o8KnwxbdmhZ/estimating-training-compute-of-deep-learning-models",0,"","scaling-laws"],["Priors, Hierarchy, and Information Asymmetry for Skill Transfer in Reinforcement Learning","Sasha Salter and 3 others","2022","paper","Published at the International Conference on Learning\n  Representations, 2023","arxiv.org/abs/2201.08115",0,"","agents"],["Safe Deep RL in 3D Environments using Human Feedback","Matthew Rahtz and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.08102",0,"","rlhf agents"],["Safety-Aware Multi-Agent Apprenticeship Learning","Junchen Zhao","2022","paper","arXiv preprint","arxiv.org/abs/2201.08111",0,"","evals agents"],["What's Up With Confusingly Pervasive Goal Directedness?","Raemon","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/DJnvFsZ2maKxPi7v7/what-s-up-with-confusingly-pervasive-goal-directedness",0,"",""],["AI acceleration from a safety perspective: Trade-offs and considerations","mariushobbhahn and Tilman","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LqWPSNRu2t46fv6hK/ai-acceleration-from-a-safety-perspective-trade-offs-and",0,"","forecasting"],["European Union AI Development and Governance Partnerships","EU AI Governance","2022","blog","EA Forum","forum.effectivealtruism.org/posts/DK7N5YofbM2cfPi8h/european-union-ai-development-and-governance-partnerships",0,"","governance"],["Improving Behavioural Cloning with Human-Driven Dynamic Dataset Augmentation","Federico Malato and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.07719",0,"","evals agents"],["My plan for a “Most Important Century” reading group","Jack O'Brien","2022","blog","EA Forum","forum.effectivealtruism.org/posts/JNcp9c7Gzt5hBwA8u/my-plan-for-a-most-important-century-reading-group",0,"",""],["Alex Turner - Will powerful AIs tend to seek┬ápower-by Towards Data Science-video_id 8afHG61YmKM-date 20220119","Alex Turner","2022","report","drive.google.com","drive.google.com/file/d/1FHuZFwyHq8cBSLrNNMW0ln_vk3uzGe4F/view?usp=share_link",0,"",""],["Clarifications about structural risk from AI","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/oqveRcMwRMDk6SYXM/clarifications-about-structural-risk-from-ai",0,"",""],["Empowerment and Stakeholder Management","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/empowerment-and-stakeholder-management/",0,"",""],["Solving Dynamic Principal-Agent Problems with a Rationally Inattentive Principal","Tong Mu and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2202.01691",0,"","agents"],["Spurious normativity enhances learning of compliance and enforcement behavior in artificial agents","Raphael Koster and 5 others","2022","blog","deepmind.com","www.deepmind.com/blog/spurious-normativity-enhances-learning-of-compliance-and-enforcement-behavior-in-artificial-agents",0,"","agents"],["The longtermist AI governance landscape: a basic overview","Sam Clarke","2022","blog","EA Forum","forum.effectivealtruism.org/posts/ydpo7LcJWhrr2GJrx/the-longtermist-ai-governance-landscape-a-basic-overview",0,"","governance"],["Challenges with Breaking into MIRI-Style Research","Chris_Leong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kcbo4rXu3jYPnauoK/challenges-with-breaking-into-miri-style-research",0,"","agents theory"],["Different way classifiers can be diverse","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/rv65vAPqpZGFLcnnD/different-way-classifiers-can-be-diverse",0,"",""],["FLI launches Worldbuilding Contest with $100,000 in prizes","ggilgallon","2022","blog","EA Forum","forum.effectivealtruism.org/posts/LjExZCPCHnNNTFDfq/fli-launches-worldbuilding-contest-with-usd100-000-in-prizes",0,"","governance policy"],["How I'm thinking about GPT-N","delton137","2022","blog","LessWrong","www.lesswrong.com/posts/iQabBACQwbWyHFKZq/how-i-m-thinking-about-gpt-n",0,"","scaling-laws"],["PIBBSS Fellowship: Bounty for Referrals & Deadline Extension","Anna_Gajdova","2022","blog","EA Forum","forum.effectivealtruism.org/posts/rNhWqG3gHgXeWLhrT/pibbss-fellowship-bounty-for-referrals-and-deadline",0,"",""],["Planning Not to Talk: Multiagent Systems that are Robust to Communication Loss","Mustafa O. Karabag and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.06619",0,"","agents policy"],["Scalar reward is not enough for aligned AGI","Peter Vamplew","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/eeEEgNeTepZb6F6NF/scalar-reward-is-not-enough-for-aligned-agi",0,"",""],["Truthful LMs as a warm-up for aligned AGI","Jacob_Hilton","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/jWkqACmDes6SoAiyE/truthful-lms-as-a-warm-up-for-aligned-agi",0,"",""],["Measuring Non-Probabilistic Uncertainty: A cognitive, logical and computational assessment of known and unknown unknowns","Florian Ellsaesser and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.05818",0,"",""],["Assorted cold-ish links","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/assorted-cold-ish-links/",0,"",""],["The Greedy Doctor Problem... turns out to be relevant to the ELK problem?","Jan","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/dDzHJGmyeQa2tGmqH/the-greedy-doctor-problem-turns-out-to-be-relevant-to-the",0,"","eliciting-latent-knowledge"],["Tools and Practices for Responsible AI Engineering","Ryan Soklaski and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.05647",0,"","evals robustness"],["EU's importance for AI governance is conditional on AI trajectories - a case study","MathiasKB","2022","blog","EA Forum","forum.effectivealtruism.org/posts/3D9bkGEtCvgQZEoAd/eu-s-importance-for-ai-governance-is-conditional-on-ai",0,"","governance policy"],["ULTRA: A Data-driven Approach for Recommending Team Formation in Response to Proposal Calls","Biplav Srivastava and 5 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.05646",0,"",""],["New year, new research agenda post","Charlie Steiner","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/zuHezdoGr2KtM2n43/new-year-new-research-agenda-post",0,"",""],["Revelation of Task Difficulty in AI-aided Education","Yitzhak Spielberg and Amos Azaria","2022","paper","arXiv preprint","arxiv.org/abs/2201.04633",0,"",""],["The Concept of Criticality in AI Safety","Yitzhak Spielberg and Amos Azaria","2022","paper","arXiv preprint","arxiv.org/abs/2201.04632",0,"","agents monitoring"],["Value extrapolation partially resolves symbol grounding","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/thZdioHTZALRPKmiH/value-extrapolation-partially-resolves-symbol-grounding",0,"",""],["An Open Philanthropy grant proposal: Causal representation learning of human preferences","PabloAMC","2022","blog","LessWrong","www.lesswrong.com/posts/5BkEoJFEqQEWy9GcL/an-open-philanthropy-grant-proposal-causal-representation",0,"","interpretability"],["Danijar Hafner - Gaming our way to┬áAGI-by Towards Data Science-video_id Bgz9eMcE5Do-date 20220112","Danijar Hafner and Jeremie Harris","2022","report","drive.google.com","drive.google.com/file/d/18mfquEOt3Ofspo_qMr2R9P8KLsfCoumJ/view?usp=share_link",0,"",""],["Future ML Systems Will Be Qualitatively Different","jsteinhardt","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/pZaPhGg2hmmPwByHc/future-ml-systems-will-be-qualitatively-different",0,"",""],["how timelines fall","Tamsin Leake","2022","blog","carado.moe","carado.moe/how-timelines-fall.html",0,"","forecasting"],["The Turing Trap: The Promise & Peril of Human-Like Artificial Intelligence","Erik Brynjolfsson","2022","paper","arXiv preprint","arxiv.org/abs/2201.04200",0,"","policy"],["Understanding the two-head strategy for teaching ML to answer questions honestly","Adam Scherlis","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ntmbm79zQakr29XLw/understanding-the-two-head-strategy-for-teaching-ml-to",0,"","eliciting-latent-knowledge"],["uploading people for alignment purposes","Tamsin Leake","2022","blog","carado.moe","carado.moe/upload-for-alignment.html",0,"",""],["What is the role of Bayesian ML for AI alignment/safety?","mariushobbhahn","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cpN8axaLbpHDix9ie/what-is-the-role-of-bayesian-ml-for-ai-alignment-safety",0,"",""],["Why it matters if \"ideas get harder to find\"","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/why-it-matters-if-ideas-get-harder-to-find/",0,"",""],["Critiquing Scasper's Definition of Subjunctive Dependence","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/aj3LycvDj3kvnW8G6/critiquing-scasper-s-definition-of-subjunctive-dependence",0,"","theory"],["The Effects of Reward Misspecification: Mapping and Mitigating Misaligned Models","Alexander Pan and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.03544",0,"","reward-hacking agents monitoring"],["Assessing Policy, Loss and Planning Combinations in Reinforcement Learning using a New Modular Architecture","Tiago Gaspar Oliveira and Arlindo L. Oliveira","2022","paper","arXiv preprint","arxiv.org/abs/2201.02874",0,"","agents policy robustness"],["Modeling Human-AI Team Decision Making","Wei Ye and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.02759",0,"","evals agents"],["How artistic ideas could get harder to find","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/how-artistic-ideas-could-get-harder-to-find/",0,"",""],["questions about the cosmos and rich computations","Tamsin Leake","2022","blog","carado.moe","carado.moe/questions-cosmos-computations.html",0,"",""],["AI alignment research links","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/xSDWS8yWWPcqAa8NR/ai-alignment-research-links-1",0,"",""],["Brain Efficiency: Much More than You Wanted to Know","jacob_cannell","2022","blog","LessWrong","www.lesswrong.com/posts/xwBuoE9p8GE7RAuhd/brain-efficiency-much-more-than-you-wanted-to-know",0,"","forecasting"],["Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","Alethea Power and 4 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.02177",0,"","robustness training-data"],["Importance of foresight evaluations within ELK","Jonathan Uesato","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/mvGNKQ6iSDf3d4gCi/importance-of-foresight-evaluations-within-elk",0,"","eliciting-latent-knowledge evals"],["Seeking beta readers","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/seeking-beta-readers/",0,"",""],["Signaling isn't about signaling, it's about Goodhart","Valentine","2022","blog","LessWrong","www.lesswrong.com/posts/dPifaJGRWMm8rKxCP/signaling-isn-t-about-signaling-it-s-about-goodhart",0,"","goodharts-law"],["A Reaction to Wolfgang Schwarz's \"On Functional Decision Theory\"","Heighn","2022","blog","LessWrong","www.lesswrong.com/posts/9yhKRuMwEqB3rQucJ/a-reaction-to-wolfgang-schwarz-s-on-functional-decision",0,"","theory"],["AI alignment research links","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/ai-alignment-research-links/",0,"",""],["brittle physics and the nature of X-risks","Tamsin Leake","2022","blog","carado.moe","carado.moe/brittle-physics.html",0,"",""],["Consider trying the ELK contest (I am)","Holden Karnofsky","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Q2BJnpNh8e6RAWFnm/consider-trying-the-elk-contest-i-am",0,"","eliciting-latent-knowledge"],["GovAI Annual Report 2021","GovAI","2022","blog","EA Forum","forum.effectivealtruism.org/posts/cecf9mGdqtxbfEt7z/govai-annual-report-2021-2",0,"","governance policy"],["Robust Self-Supervised Audio-Visual Speech Recognition","Bowen Shi and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.01763",0,"","benchmarks robustness"],["2021-22 New Year review","Victoria Krakovna","2022","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2022/01/04/2021-22-new-year-review/",0,"",""],["China’s New AI Governance Initiatives Shouldn’t Be Ignored","Matt Sheehan","2022","report","carnegieendowment.org","carnegieendowment.org/2022/01/04/china-s-new-ai-governance-initiatives-shouldn-t-be-ignored-pub-86127",0,"","governance"],["More Is Different for AI","jsteinhardt","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Lp4Q9kSGsJHLfoHX3/more-is-different-for-ai",0,"",""],["Promising posts on AF that have fallen through the cracks","Evan R. Murphy","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/WerwgmeYZYGC2hKXN/promising-posts-on-af-that-have-fallen-through-the-cracks",0,"",""],["Where's Today's Beethoven?","Holden Karnofsky","2022","blog","cold-takes.com","www.cold-takes.com/wheres-todays-beethoven/",0,"",""],["Apply for research internships at ARC!","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/BRsxztzkTzScFQfDW/apply-for-research-internships-at-arc",0,"",""],["Execute Order 66: Targeted Data Poisoning for Reinforcement Learning","Harrison Foley and 3 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.00762",0,"","agents policy training-data"],["Have I done enough planning or should I plan more?","Ruiqi He and 2 others","2022","paper","arXiv preprint","arxiv.org/abs/2201.00764",0,"","deception policy"],["How an alien theory of mind might be unlearnable","Stuart_Armstrong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/kMJxwCZ4mc9w4ezbs/how-an-alien-theory-of-mind-might-be-unlearnable",0,"",""],["Prizes for ELK proposals","paulfchristiano","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/QEYWkRoCn4fZxXQAY/prizes-for-elk-proposals",0,"","eliciting-latent-knowledge"],["\"The Scaling Hypothesis\"","Gwern Branwen","2022","blog","gwern.net","www.gwern.net/Scaling-hypothesis.page",0,"",""],["$1000 USD prize - Circular Dependency of Counterfactuals","Chris_Leong","2022","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gzw6FwPD9FeL4GTWC/usd1000-usd-prize-circular-dependency-of-counterfactuals",0,"","theory"],["December 2021 Newsletter","Rob Bensinger","2022","blog","intelligence.org","intelligence.org/2021/12/31/december-2021-newsletter/",0,"",""],["Locating and Editing Factual Associations in GPT","Kevin Meng and 3 others","2022","report","rome.baulab.info","rome.baulab.info/",0,"",""],["Will the EU regulations on AI matter to the rest of the world?","anonymous","2022","blog","EA Forum","forum.effectivealtruism.org/posts/Kd759DsB68t8wxCES/will-the-eu-regulations-on-ai-matter-to-the-rest-of-the",0,"","evals governance policy"],["Counterexamples to some ELK proposals","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FnZws8NuKw6BJzmvZ/counterexamples-to-some-elk-proposals",0,"","eliciting-latent-knowledge"],["Eliciting Latent Knowledge Via Hypothetical Sensors","John_Maxwell","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/H7v5yyXAmmgu9DJmi/eliciting-latent-knowledge-via-hypothetical-sensors",0,"","eliciting-latent-knowledge"],["We need a theory of anthropic measure binding","mako yass","2021","blog","LessWrong","www.lesswrong.com/posts/gx6GEnpLkTXn3NFSS/we-need-a-theory-of-anthropic-measure-binding",0,"","theory"],["Increased Availability and Willingness for Deployment of Resources for Effective Altruism and Long-Termism","Evan_Gaensbauer","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ekComvhb2HREowgah/increased-availability-and-willingness-for-deployment-of",0,"",""],["Reverse-engineering using interpretability","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qrn2dRSwNratuM3tq/reverse-engineering-using-interpretability",0,"","interpretability"],["Gradient Hacking via Schelling Goals","Adam Scherlis","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/A9eAPjpFjPwNW2rku/gradient-hacking-via-schelling-goals",0,"",""],["What counts as death?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/what-counts-as-death/",0,"",""],["13 Very Different Stances on AGI","Ozzie Gooen","2021","blog","EA Forum","forum.effectivealtruism.org/posts/SZFDtA4pjZzepdacv/13-very-different-stances-on-agi",0,"","forecasting"],["AGI alignment results from a series of aligned actions","anonymous","2021","blog","EA Forum","forum.effectivealtruism.org/posts/qrapebbppHASWB3W9/agi-alignment-results-from-a-series-of-aligned-actions",0,"","governance forecasting"],["less quantum immortality? • carado.moe","Tamsin Leake","2021","blog","carado.moe","carado.moe/less-quantum-immortality.html",0,"",""],["Why don't governments seem to mind that companies are explicitly trying to make AGIs?","ozziegooen","2021","blog","LessWrong","www.lesswrong.com/posts/bbjeShykGbTwFsiMu/why-don-t-governments-seem-to-mind-that-companies-are",0,"","governance"],["database transactions: you guessed it, it's WASM again","Tamsin Leake","2021","blog","carado.moe","carado.moe/database-transactions-wasm.html",0,"",""],["My Overview of the AI Alignment Landscape: Threat Models","Neel Nanda","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3DFBbPFZyscrAiTKS/my-overview-of-the-ai-alignment-landscape-threat-models",0,"","goodharts-law power-seeking"],["non-scarce compute means moral patients might not get optimized out","Tamsin Leake","2021","blog","carado.moe","carado.moe/nonscarce-compute-optimize-out.html",0,"",""],["thinking about psi: as a more general json","Tamsin Leake","2021","blog","carado.moe","carado.moe/psi-json.html",0,"",""],["yes room above paperclips?","Tamsin Leake","2021","blog","carado.moe","carado.moe/above-paperclips-2.html",0,"",""],["Risks from AI persuasion","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5cWtwATHL6KyzChck/risks-from-ai-persuasion",0,"",""],["Should the EA community have a DL engineering fellowship?","PabloAMC","2021","blog","EA Forum","forum.effectivealtruism.org/posts/orpsrpYDMAdWRPFZW/should-the-ea-community-have-a-dl-engineering-fellowship",0,"",""],["2021 AI Alignment Literature Review and Charity Comparison","Larks","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/C4tR3BEpuWviT7Sje/2021-ai-alignment-literature-review-and-charity-comparison",0,"",""],["Free Guy, a rom-com on the moral patienthood of digital sentience","mic","2021","blog","EA Forum","forum.effectivealtruism.org/posts/haczkGjMusjyozTiH/free-guy-a-rom-com-on-the-moral-patienthood-of-digital-1",0,"","robustness"],["Reply to Eliezer on Biological Anchors","HoldenKarnofsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nNqXfnjiezYukiMJi/reply-to-eliezer-on-biological-anchors",0,"",""],["Utopia links","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/utopia-links/",0,"",""],["Why don't governments seem to mind that companies are explicitly trying to make AGIs?","Ozzie Gooen","2021","blog","EA Forum","forum.effectivealtruism.org/posts/wdk3LCg6iFxknCYG4/why-don-t-governments-seem-to-mind-that-companies-are",0,"","governance forecasting"],["Worst-case thinking in AI alignment","Buck","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yTvBSFrXhZfL8vr5a/worst-case-thinking-in-ai-alignment",0,"",""],["A Mathematical Framework for Transformer Circuits","Nelson Elhage and 24 others","2021","blog","transformer-circuits.pub","transformer-circuits.pub/2021/framework/index.html",0,"","mechanistic-interpretability"],["Bet with Zvi about Omicron","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/bet-with-zvi-about-omicron/",0,"",""],["Potential gears level explanations of smooth progress","ryan_greenblatt","2021","blog","LessWrong","www.lesswrong.com/posts/ShrAZXjTs5HTxDmGM/potential-gears-level-explanations-of-smooth-progress",0,"","forecasting"],["Transformer Circuits","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2269iGRnWruLHsZ5r/transformer-circuits",0,"","interpretability mechanistic-interpretability"],["Worldbuilding exercise: The Highwayverse.","Yair Halberstadt","2021","blog","LessWrong","www.lesswrong.com/posts/fEw4KmqjgAKDWqwYM/worldbuilding-exercise-the-highwayverse",0,"","theory"],["Bayesian Mindset","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/the-bayesian-mindset/",0,"",""],["DB-BERT: a Database Tuning Tool that \"Reads the Manual\"","Immanuel Trummer","2021","paper","arXiv preprint","arxiv.org/abs/2112.10925",0,"","evals benchmarks"],["Demanding and Designing Aligned Cognitive Architectures","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cDR8GkzCaxXoovPwh/demanding-and-designing-aligned-cognitive-architectures",0,"","governance"],["Introducing a New Course on the Economics of AI","akorinek","2021","blog","EA Forum","forum.effectivealtruism.org/posts/FJsk9i9c9zLC7eKLF/introducing-a-new-course-on-the-economics-of-ai",0,"","governance"],["Researcher incentives cause smoother progress on benchmarks","ryan_greenblatt","2021","blog","LessWrong","www.lesswrong.com/posts/3b79GzkPXLfHyxRhv/researcher-incentives-cause-smoother-progress-on-benchmarks",0,"","benchmarks forecasting"],["Demanding and Designing Aligned Cognitive Architectures","Koen Holtman","2021","paper","arXiv preprint","arxiv.org/abs/2112.10190",0,"","policy"],["Don't Influence the Influencers!","lhc","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Eg9FE2iYp3ngySsMD/don-t-influence-the-influencers",0,"",""],["[Extended Deadline: Jan 23rd] Announcing the PIBBSS Summer Research Fellowship","nora","2021","blog","EA Forum","forum.effectivealtruism.org/posts/hhLbFmhXbmaX5PcCa/extended-deadline-jan-23rd-announcing-the-pibbss-summer",0,"",""],["Disentangling Perspectives On Strategy-Stealing in AI Safety","shawnghu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CtGwGgxfoefiwfcor/disentangling-perspectives-on-strategy-stealing-in-ai-safety",0,"",""],["Exploring Decision Theories With Counterfactuals and Dynamic Agent Self-Pointers","JoshuaOSHickman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zupqBxrNKpT5dhFQb/exploring-decision-theories-with-counterfactuals-and-dynamic",0,"","agents theory"],["Introducing the Principles of Intelligent Behaviour in Biological and Social Systems (PIBBSS) Fellowship","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/4Tjz4EJ8DozE9z5nQ/introducing-the-principles-of-intelligent-behaviour-in",0,"",""],["Cold Links: misc","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-misc/",0,"",""],["WebGPT: Browser-assisted question-answering with human feedback","Reiichiro Nakano and 17 others","2021","paper","arXiv preprint","arxiv.org/abs/2112.09332",0,"","rlhf evals"],["Elicitation for Modeling Transformative AI Risks","Davidmanheim","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kz9NHBMeJxzSwb7R9/elicitation-for-modeling-transformative-ai-risks",0,"",""],["Evidence Sets: Towards Inductive-Biases based Analysis of Prosaic AGI","bayesian_kitten","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/aW5CPtqtvs2EYKMMK/evidence-sets-towards-inductive-biases-based-analysis-of",0,"","robustness"],["Housing Markets, Satisficers, and One-Track Goodhart","Jemist","2021","blog","LessWrong","www.lesswrong.com/posts/Rypb63HLXWYw8CdSr/housing-markets-satisficers-and-one-track-goodhart",0,"","goodharts-law"],["Motivations, Natural Selection, and Curriculum Engineering","Oliver Sourbut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RwYh4grJs4pbJdTh3/motivations-natural-selection-and-curriculum-engineering",0,"",""],["Reviews of \"Is power-seeking AI an existential risk?\"","Joe_Carlsmith","2021","blog","EA Forum","forum.effectivealtruism.org/posts/GRv3KB2nPFRREXb5o/reviews-of-is-power-seeking-ai-an-existential-risk",0,"","red-teaming power-seeking"],["Reviews of “Is power-seeking AI an existential risk?”","Joe Carlsmith","2021","blog","LessWrong","www.lesswrong.com/posts/qRSgHLb8yLXzDg4nf/reviews-of-is-power-seeking-ai-an-existential-risk",0,"","power-seeking"],["Universality and the “Filter”","maggiehayes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/R67qpBj5doGcjQzmc/universality-and-the-filter",0,"",""],["AI Safety: Applying to Graduate Studies","frances_lorenz","2021","blog","EA Forum","forum.effectivealtruism.org/posts/KHQKbwWk7oosAxnMC/ai-safety-applying-to-graduate-studies",0,"","deception"],["Framing approaches to alignment and the hard problem of AI cognition","ryan_greenblatt","2021","blog","LessWrong","www.lesswrong.com/posts/Cj6PBGSjqkPfXbwCF/framing-approaches-to-alignment-and-the-hard-problem-of-ai",0,"",""],["My Overview of the AI Alignment Landscape: A Bird's Eye View","Neel Nanda","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SQ9cZtfrzDJmw9A2m/my-overview-of-the-ai-alignment-landscape-a-bird-s-eye-view",0,"","scalable-oversight interpretability forecasting"],["My Overview of the AI Alignment Landscape: A Bird’s Eye View","Neel Nanda","2021","blog","EA Forum","forum.effectivealtruism.org/posts/hurNCKfoYacJ5PSod/my-overview-of-the-ai-alignment-landscape-a-bird-s-eye-view",0,"",""],["Ngo’s view on alignment difficulty","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/12/14/ngos-view-on-alignment-difficulty/",0,"",""],["ARC is hiring!","paulfchristiano and Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dLoK6KGcHAoudtwdo/arc-is-hiring",0,"",""],["ARC's first technical report: Eliciting Latent Knowledge","paulfchristiano and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qHCDysDnvhteW7kRd/arc-s-first-technical-report-eliciting-latent-knowledge",0,"","eliciting-latent-knowledge"],["Consequentialism & corrigibility","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/KDMLJEXTWtkZWheXt/consequentialism-and-corrigibility",0,"",""],["Decision Theory Breakdown—Personal Attempt at a Review","Jake Arft-Guatelli","2021","blog","LessWrong","www.lesswrong.com/posts/nkKAYBgG9GXJHm2hE/decision-theory-breakdown-personal-attempt-at-a-review",0,"","theory"],["Filling gaps in trustworthy development of AI","Shahar Avin and 11 others","2021","paper","Science (2021) Vol 374, Issue 6573, pp. 1327-1329","arxiv.org/abs/2112.07773",0,"","governance"],["Interlude: Agents as Automobiles","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cxkwQmys6mCB6bjDA/interlude-agents-as-automobiles",0,"","agents"],["Ngo's view on alignment difficulty","Richard_Ngo and Eliezer Yudkowsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gf9hhmSvpZfyfS34B/ngo-s-view-on-alignment-difficulty",0,"","governance"],["Programmatic Reward Design by Example","Weichao Zhou and Wenchao Li","2021","paper","arXiv preprint","arxiv.org/abs/2112.08438",0,"","interpretability agents"],["Should we rely on the speed prior for safety?","Marc-Everin Carauleanu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/iALu99gYbodt4mLqg/should-we-rely-on-the-speed-prior-for-safety",0,"",""],["The Natural Abstraction Hypothesis: Implications and Evidence","TheMcDouglas","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fut8dtFsBYRz8atFF/the-natural-abstraction-hypothesis-implications-and-evidence",0,"","interpretability"],["Visualizing Utopia","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/visualizing-utopia/",0,"",""],["Hard-Coding Neural Computation","MadHatter","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HkghiK6Rt35nbgwKA/hard-coding-neural-computation",0,"","interpretability"],["Language Model Alignment Research Internships","Ethan Perez","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vHcGGrnzcshybrCJD/language-model-alignment-research-internships",0,"",""],["Solving Interpretability Week","Logan Riggs","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jbdDxmhxBygDqbQMD/solving-interpretability-week",0,"","interpretability"],["Stackelberg Games and Cooperative Commitment: My Thoughts and Reflections on a 2-Month Research Project","Ben Bucknall","2021","blog","EA Forum","forum.effectivealtruism.org/posts/NzA9D9m823Tx2msm3/stackelberg-games-and-cooperative-commitment-my-thoughts-and",0,"",""],["Summary of the Acausal Attack Issue for AIXI","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YbahERfcjTu7LZNQ6/summary-of-the-acausal-attack-issue-for-aixi",0,"",""],["Understanding and controlling auto-induced distributional shift","LRudL","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rTYGMbmEsFkxyyXuR/understanding-and-controlling-auto-induced-distributional",0,"",""],["What’s the backward-forward FLOP ratio for Neural Networks?","Marius Hobbhahn and Jsevillamol","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fnjKpBoWJXcSDwhZk/what-s-the-backward-forward-flop-ratio-for-neural-networks",0,"","forecasting"],["Redwood's Technique-Focused Epistemic Strategy","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2xrBxhRhde7Xddt38/redwood-s-technique-focused-epistemic-strategy",0,"","robustness"],["Some abstract, non-technical reasons to be non-maximally-pessimistic about AI alignment","Rob Bensinger","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vT4tsttHgYJBoKi4n/some-abstract-non-technical-reasons-to-be-non-maximally",0,"",""],["Teaser: Hard-coding Transformer Models","MadHatter","2021","blog","LessWrong","www.lesswrong.com/posts/Lq6jo5j9ty4sezT7r/teaser-hard-coding-transformer-models",0,"","interpretability"],["Moore's Law, AI, and the pace of progress","Veedrac","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/aNAFrGbzXddQBMDqh/moore-s-law-ai-and-the-pace-of-progress",0,"",""],["Transforming myopic optimization to ordinary optimization - Do we want to seek convergence for myopic optimization problems?","tailcalled","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ayt24gxcjfY3tDmwK/transforming-myopic-optimization-to-ordinary-optimization-do",0,"",""],["What role should evolutionary analogies play in understanding AI takeoff speeds?","anson","2021","blog","EA Forum","forum.effectivealtruism.org/posts/aSDnzAm85a3Pi87rm/what-role-should-evolutionary-analogies-play-in",0,"","forecasting"],["What role should evolutionary analogies play in understanding AI takeoff speeds?","anson.ho","2021","blog","LessWrong","www.lesswrong.com/posts/teD4xjwoeWc4LyRAD/what-role-should-evolutionary-analogies-play-in",0,"","forecasting"],["Enabling more feedback","JJ Hepburn","2021","blog","EA Forum","forum.effectivealtruism.org/posts/6qaaRnu6oN4pdAnWF/enabling-more-feedback",0,"",""],["The Plan","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3L46WGauGpr7nYubu/the-plan",0,"",""],["The Promise and Peril of Finite Sets","davidad","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qLaShfcnXGnYeKFJW/the-promise-and-peril-of-finite-sets",0,"",""],["There is essentially one best-validated theory of cognition.","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/NB9QrBa335GDijuyn/there-is-essentially-one-best-validated-theory-of-cognition",0,"",""],["TV shows I wish I could watch: Intergalactic Immigration Wars","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/tv-shows-i-wish-i-could-watch-intergalactic-immigration-wars/",0,"",""],["Understanding Gradient Hacking","peterbarnett","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bdayaswyewjxxrQmB/understanding-gradient-hacking",0,"",""],["[MLSN #2]: Adversarial Training","Dan H","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7GQZyooNi5nqgoyyJ/mlsn-2-adversarial-training",0,"",""],["Conversation on technology forecasting and gradualism","Richard_Ngo and 3 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nPauymrHwpoNr6ipx/conversation-on-technology-forecasting-and-gradualism",0,"","forecasting"],["Conversation on technology forecasting and gradualism","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/12/09/conversation-on-technology-forecasting-and-gradualism/",0,"","forecasting"],["emotionally appreciating grand political visions","Tamsin Leake","2021","blog","carado.moe","carado.moe/appreciating-grand-political-visions.html",0,"",""],["EU AI Act now has a section on general purpose AI systems","MathiasKB","2021","blog","EA Forum","forum.effectivealtruism.org/posts/3itL9GJcxvQC5Pp5D/eu-ai-act-now-has-a-section-on-general-purpose-ai-systems",0,"","governance"],["freedom and diversity in Albion's Seed","Tamsin Leake","2021","blog","carado.moe","carado.moe/albions-seed.html",0,"",""],["Introduction to inaccessible information","Ryan Kidd","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CYKeDjD7FEvAnzBBF/introduction-to-inaccessible-information",0,"","interpretability"],["ML Safety Newsletter #2","Dan Hendrycks","2021","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-2",0,"",""],["non-interfering superintelligence and remaining philosophical progress: a deterministic utopia","Tamsin Leake","2021","blog","carado.moe","carado.moe/noninterf-superint.html",0,"",""],["PixMix: Dreamlike Pictures Comprehensively Improve Safety Measures","Dan Hendrycks and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2112.05135",0,"","robustness monitoring"],["psi rewriting","Tamsin Leake","2021","blog","carado.moe","carado.moe/psi-rewriting.html",0,"",""],["Supervised learning and self-modeling: What's \"superhuman?\"","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pz84sQKsgg3GBHQpd/supervised-learning-and-self-modeling-what-s-superhuman",0,"",""],["unoptimal superintelligence doesn't lose","Tamsin Leake","2021","blog","carado.moe","carado.moe/unoptimal-superint-doesnt-lose.html",0,"",""],["[AN #170]: Analyzing the argument for risk from power-seeking AI","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PC6QavgNDQjHbutAq/an-170-analyzing-the-argument-for-risk-from-power-seeking-ai",0,"","power-seeking"],["Creating Interactive Agents with Imitation Learning","Josh Abramson and 24 others","2021","blog","deepmind.com","www.deepmind.com/blog/creating-interactive-agents-with-imitation-learning",0,"","agents"],["Finding the multiple ground truths of CoinRun and image classification","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oCWk8QpjgyqbFHKtK/finding-the-multiple-ground-truths-of-coinrun-and-image",0,"",""],["Improving language models by retrieving from trillions of tokens","Sebastian Borgeaud and 3 others","2021","blog","deepmind.com","www.deepmind.com/blog/improving-language-models-by-retrieving-from-trillions-of-tokens",0,"",""],["Some thoughts on why adversarial training might be useful","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gi8HPM8iYZcdAEteJ/some-thoughts-on-why-adversarial-training-might-be-useful",0,"",""],["Considerations on interaction between AI and expected value of the future","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Dr3owdPqEAFK4pq8S/considerations-on-interaction-between-ai-and-expected-value",0,"",""],["Exterminating humans might be on the to-do list of a Friendly AI","RomanS","2021","blog","LessWrong","www.lesswrong.com/posts/DKmXnD8fA6NjnRWeg/exterminating-humans-might-be-on-the-to-do-list-of-a",0,"",""],["HIRING: Inform and shape a new project on AI safety at Partnership on AI","madhu_lika","2021","blog","LessWrong","www.lesswrong.com/posts/sEeh6tWpSvLSpoaH8/hiring-inform-and-shape-a-new-project-on-ai-safety-at-3",0,"","governance"],["MESA: Offline Meta-RL for Safe Adaptation and Fault Tolerance","Michael Luo and 8 others","2021","paper","Workshop on Safe and Robust Control of Uncertain Systems at the\n  35th Conference on Neural Information Processing Systems (NeurIPS 2021),\n  Online","arxiv.org/abs/2112.03575",0,"",""],["More Christiano, Cotra, and Yudkowsky on AI progress","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/12/06/more-christiano-cotra-and-yudkowsky-on-ai-progress/",0,"",""],["Theoretical Neuroscience For Alignment Theory","Cameron Berg","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZJY3eotLdfBPCLP3z/theoretical-neuroscience-for-alignment-theory",0,"",""],["Why Describing Utopia Goes Badly","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/why-describing-utopia-goes-badly/",0,"",""],["A Framework to Explain Bayesian Models","Jsevillamol","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SPfZiEwHotPncJBLz/a-framework-to-explain-bayesian-models",0,"",""],["A Possible Resolution To Spurious Counterfactuals","JoshuaOSHickman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/TnkDtTAqCGetvLsgr/a-possible-resolution-to-spurious-counterfactuals",0,"","theory"],["Are there alternative to solving value transfer and extrapolation?","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DjTKMEwRqpuKkJzTo/are-there-alternative-to-solving-value-transfer-and",0,"",""],["Candidate for “highest-stakes question of the next several months” (rare hot take)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/candidate-for-highest-stakes-question-of-the-next-several-months-rare-hot-take/",0,"",""],["Contribute by facilitating the AGI Safety Fundamentals Programme","Jamie Bernardi","2021","blog","EA Forum","forum.effectivealtruism.org/posts/WtwMy69JKZeHEvykc/contribute-by-facilitating-the-agi-safety-fundamentals",0,"","governance"],["Declustering, reclustering, and filling in thingspace","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/WikzbCsFjpLTRQmXn/declustering-reclustering-and-filling-in-thingspace",0,"",""],["Do neural networks learn human concepts?","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/do-neural-networks-learn-human-concepts/",0,"",""],["Information bottleneck for counterfactual corrigibility","tailcalled","2021","blog","LessWrong","www.lesswrong.com/posts/wpmdkftNZy26z4MTW/information-bottleneck-for-counterfactual-corrigibility",0,"",""],["ML Alignment Theory Program under Evan Hubinger","ozhang and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FpokmCnbP3CEZ5h4t/ml-alignment-theory-program-under-evan-hubinger",0,"",""],["Modeling Failure Modes of High-Level Machine Intelligence","Ben Cottier and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3Eq5Rq5uQ97kt8B8f/modeling-failure-modes-of-high-level-machine-intelligence",0,"",""],["More Christiano, Cotra, and Yudkowsky on AI progress","Eliezer Yudkowsky and Ajeya Cotra","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fS7Zdj2e2xMqE6qja/more-christiano-cotra-and-yudkowsky-on-ai-progress",0,"","forecasting"],["Retrospective on the Summer 2021 AGI Safety Fundamentals","Dewi","2021","blog","EA Forum","forum.effectivealtruism.org/posts/QNhpbvyAHZwBiyKmB/retrospective-on-the-summer-2021-agi-safety-fundamentals",0,"",""],["Are limited-horizon agents a good heuristic for the off-switch problem?","anonymous","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/itTLCFj5NCHhFbK2Q/are-limited-horizon-agents-a-good-heuristic-for-the-off",0,"","agents robustness"],["Behavior Cloning is Miscalibrated","leogao","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BgoKdAzogxmgkuuAt/behavior-cloning-is-miscalibrated",0,"",""],["Interpreting Yudkowsky on Deep vs Shallow Knowledge","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GSBCw94DsxLgDat6r/interpreting-yudkowsky-on-deep-vs-shallow-knowledge",0,"",""],["the deobfuscation conjecture","Tamsin Leake","2021","blog","carado.moe","carado.moe/deobfuscation-conjecture.html",0,"",""],["Agency: What it is and why it matters","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qJBkcGW4GitfQ4BBy/agency-what-it-is-and-why-it-matters",0,"",""],["Agents as P₂B Chain Reactions","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oiftkZnFBqyHGALwv/agents-as-p-b-chain-reactions",0,"","agents"],["Misc. questions about EfficientZero","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rtBpBNgXjwtsLJDbG/misc-questions-about-efficientzero-1",0,"",""],["Shulman and Yudkowsky on AI progress","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/12/04/shulman-and-yudkowsky-on-ai-progress/",0,"",""],["Shulman and Yudkowsky on AI progress","CarlShulman and EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/brhX6axLaqxtDKWXe/shulman-and-yudkowsky-on-ai-progress",0,"","forecasting"],["think in what ?","Tamsin Leake","2021","blog","carado.moe","carado.moe/think-in-what.html",0,"",""],["Voting Theory has a HOLE","Anthony Repetto","2021","blog","EA Forum","forum.effectivealtruism.org/posts/zyXtnrokGyR7yhqsg/voting-theory-has-a-hole",0,"","policy"],["$100/$50 rewards for good references","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ex2qcux8TQXigGAfv/usd100-usd50-rewards-for-good-references",0,"","robustness"],["[Linkpost] A General Language Assistant as a Laboratory for Alignment","Quintin Pope","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dktT3BiinsBZLw96h/linkpost-a-general-language-assistant-as-a-laboratory-for",0,"",""],["Biology-Inspired AGI Timelines: The Trick That Never Works","Eliezer Yudkowsky","2021","blog","intelligence.org","intelligence.org/2021/12/03/biology-inspired-agi-timelines-the-trick-that-never-works/",0,"","forecasting"],["EA megaprojects continued","mariushobbhahn and 4 others","2021","blog","EA Forum","forum.effectivealtruism.org/posts/faezoENQwSTyw9iop/ea-megaprojects-continued",0,"","forecasting"],["Formalizing Policy-Modification Corrigibility","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RAnb2A5vML95rBMyd/formalizing-policy-modification-corrigibility",0,"","policy"],["Shulman and Yudkowsky on AI progress","Eliezer Yudkowsky and CarlShulman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/sCCdCLPN9E3YvdZhj/shulman-and-yudkowsky-on-ai-progress",0,"","forecasting"],["What “defense layers” should governments, AI labs, and businesses use to prevent catastrophic AI failures?","alexlintz","2021","blog","EA Forum","forum.effectivealtruism.org/posts/FHAAJKTFd92YmaTHc/what-defense-layers-should-governments-ai-labs-and",0,"","governance"],["AXRP Episode 12 - AI Existential Risk with Paul Christiano","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/krsjmpDB4kgDq6pdu/axrp-episode-12-ai-existential-risk-with-paul-christiano",0,"",""],["Morality is Scary","Wei Dai","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/y5jAuKqkShdjMNZab/morality-is-scary",0,"",""],["No need to click","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/no-need-to-click/",0,"",""],["Sydney AI Safety Fellowship","Chris_Leong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jotEekAQxmrwcMf9e/sydney-ai-safety-fellowship",0,"",""],["Sydney AI Safety Fellowship","Chris Leong","2021","blog","EA Forum","forum.effectivealtruism.org/posts/QrwnajRpteBZhQZnu/sydney-ai-safety-fellowship",0,"",""],["A General Language Assistant as a Laboratory for Alignment","Amanda Askell and 21 others","2021","paper","arXiv preprint","arxiv.org/abs/2112.00861",0,"","evals"],["Biology-Inspired AGI Timelines: The Trick That Never Works","Eliezer Yudkowsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ax695frGJEzGxFBK4/biology-inspired-agi-timelines-the-trick-that-never-works",0,"","forecasting"],["Certified Adversarial Defenses Meet Out-of-Distribution Corruptions: Benchmarking Robustness and Simple Baselines","Jiachen Sun and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2112.00659",0,"","evals benchmarks assurance robustness training-data"],["CHAI Newsletter #3 2021","CHAI","2021","report","drive.google.com","drive.google.com/file/d/15notk1PoUa8YWFONRbZhAE_ljUD6_oZJ/view?usp=sharing",0,"",""],["Hypotheses about Finding Knowledge and One-Shot Causal Entanglements","Jemist","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YKBbqMXSKetTQ2HBW/hypotheses-about-finding-knowledge-and-one-shot-causal",0,"",""],["On the Expressivity of Markov Reward","David Abel and 6 others","2021","blog","deepmind.com","www.deepmind.com/blog/on-the-expressivity-of-markov-reward",0,"",""],["AI Governance Fundamentals - Curriculum and Application","Mauricio","2021","blog","LessWrong","www.lesswrong.com/posts/hSc4yMamMzrHfJrKF/ai-governance-fundamentals-curriculum-and-application",0,"","governance"],["Did life get better during the pre-industrial era? (Ehhhh)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/did-life-get-better-during-the-pre-industrial-era-ehhhh/",0,"",""],["From AI for People to AI for the World and the Universe","Seth Baum and Andrea Owe","2021","report","gcrinstitute.org","gcrinstitute.org/papers/061_ai-world-universe.pdf",0,"",""],["Infra-Bayesian physicalism: a formal theory of naturalized induction","Vanessa Kosoy","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gHgs2e2J5azvGFatb/infra-bayesian-physicalism-a-formal-theory-of-naturalized",0,"",""],["Infra-Bayesian physicalism: proofs part I","Vanessa Kosoy","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cj3PRu8QoFm4BA8oc/infra-bayesian-physicalism-proofs-part-i",0,"",""],["Infra-Bayesian physicalism: proofs part II","Vanessa Kosoy","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CPr8bRGekTyvh7nGC/infra-bayesian-physicalism-proofs-part-ii",0,"",""],["My take on higher-order game theory","Nisan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cw84NJXAma85AmpnH/my-take-on-higher-order-game-theory",0,"",""],["Pyramid Adversarial Training Improves ViT Performance","Charles Herrmann and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.15121",0,"","robustness"],["Visible Thoughts Project and Bounty Announcement","So8res","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zRn6cLtxyNodudzhw/visible-thoughts-project-and-bounty-announcement",0,"",""],["Visible Thoughts Project and Bounty Announcement","Nate Soares","2021","blog","intelligence.org","intelligence.org/2021/11/29/visible-thoughts-project-and-bounty-announcement/",0,"",""],["AI Governance Course - Curriculum and Application","Mauricio","2021","blog","EA Forum","forum.effectivealtruism.org/posts/68ANc8KhEn6sbQ3P9/ai-governance-course-curriculum-and-application",0,"","governance"],["Comments on Allan Dafoe on AI Governance","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rjvLpRzd8mqDyZmcF/comments-on-allan-dafoe-on-ai-governance",0,"","governance"],["How to measure FLOP/s for Neural Networks empirically?","Marius Hobbhahn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jJApGWG95495pYM7C/how-to-measure-flop-s-for-neural-networks-empirically",0,"","scaling-laws"],["OOD-CV: A Benchmark for Robustness to Out-of-Distribution Shifts of Individual Nuisances in Natural Images","Bingchen Zhao and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.14341",0,"","benchmarks robustness"],["Question/Issue with the 5/10 Problem","acgt","2021","blog","LessWrong","www.lesswrong.com/posts/aAMLFc9AogbJwN8xZ/question-issue-with-the-5-10-problem",0,"","theory"],["Redwood Research is hiring for several roles","Jack R and billzito","2021","blog","EA Forum","forum.effectivealtruism.org/posts/JZqbNjtAqieivTG7Q/redwood-research-is-hiring-for-several-roles",0,"","robustness"],["Soares, Tallinn, and Yudkowsky discuss AGI cognition","So8res and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oKYWbXioKaANATxKY/soares-tallinn-and-yudkowsky-discuss-agi-cognition",0,"","forecasting"],["Soares, Tallinn, and Yudkowsky discuss AGI cognition","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/29/soares-tallinn-and-yudkowsky-discuss-agi-cognition/",0,"",""],["Soares, Tallinn, and Yudkowsky discuss AGI cognition","EliezerYudkowsky and So8res","2021","blog","EA Forum","forum.effectivealtruism.org/posts/iQrbKmJRHBjipJMh7/soares-tallinn-and-yudkowsky-discuss-agi-cognition",0,"",""],["Weighing the Milky Way and Andromeda with Artificial Intelligence","Pablo Villanueva-Domingo and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.14874",0,"",""],["Compute Research Questions and Metrics - Transformative AI and Compute [4/4]","lennart","2021","blog","LessWrong","www.lesswrong.com/posts/G4KHuYC3pHry6yMhi/compute-research-questions-and-metrics-transformative-ai-and",0,"","forecasting scaling-laws"],["Solve Corrigibility Week","Logan Riggs","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Lv3emECEjkCSHG7L7/solve-corrigibility-week",0,"",""],["The Terminology of Artificial Sentience","Janet Pauketat","2021","blog","EA Forum","forum.effectivealtruism.org/posts/HpLzFjr8PePrDpmCZ/the-terminology-of-artificial-sentience-1",0,"","governance"],["Learning from learning machines: a new generation of AI technology to meet the needs of science","Luca Pion-Tonachini and 35 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.13786",0,"",""],["Normative Disagreement as a Challenge for Cooperative AI","Julian Stastny and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.13872",0,"","agents robustness"],["AI and the Everything in the Whole Wide World Benchmark","Inioluwa Deborah Raji and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.15366",0,"","benchmarks robustness"],["EfficientZero: How It Works","1a3orn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/mRwJce3npmzbKfxws/efficientzero-how-it-works",0,"",""],["larger language models may disappoint you [or, an eternally unfinished draft]","nostalgebraist","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pv7Qpu8WSge8NRbpB/larger-language-models-may-disappoint-you-or-an-eternally",0,"",""],["Machines & Influence: An Information Systems Lens","Shashank Yadav","2021","paper","arXiv preprint","arxiv.org/abs/2111.13365",0,"","policy"],["Sentience Institute 2021 End of Year Summary","Ali","2021","blog","EA Forum","forum.effectivealtruism.org/posts/7jdEqubznyiNnY4Tn/sentience-institute-2021-end-of-year-summary-1",0,"",""],["Christiano, Cotra, and Yudkowsky on AI progress","Eliezer Yudkowsky and Ajeya Cotra","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7MCqRnZzvszsxgtJi/christiano-cotra-and-yudkowsky-on-ai-progress",0,"","forecasting"],["Christiano, Cotra, and Yudkowsky on AI progress","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/25/christiano-cotra-and-yudkowsky-on-ai-progress/",0,"",""],["Christiano, Cotra, and Yudkowsky on AI progress","Ajeya and EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ZpTEJPgvGa9ff9AcK/christiano-cotra-and-yudkowsky-on-ai-progress",0,"","governance forecasting"],["[AN #169]: Collaborating with humans without human data","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BT7yLvSdyuerqzPpc/an-169-collaborating-with-humans-without-human-data",0,"",""],["Artificial Intelligence and the Problem of Control","Stuart Russell","2021","report","dighum.ec.tuwien.ac.at","dighum.ec.tuwien.ac.at/perspectives-on-digital-humanism/artificial-intelligence-and-the-problem-of-control/",0,"",""],["HIRING: Inform and shape a new project on AI safety at Partnership on AI","Madhulika Srikumar","2021","blog","EA Forum","forum.effectivealtruism.org/posts/DcxHhLuKDWeASxGz3/hiring-inform-and-shape-a-new-project-on-ai-safety-at-1",0,"","governance"],["HIRING: Inform and shape a new project on AI safety at Partnership on AI","Madhulika Srikumar","2021","blog","LessWrong","www.lesswrong.com/posts/RCeAPWsPsKwgefFfL/hiring-inform-and-shape-a-new-project-on-ai-safety-at-1",0,"","forecasting"],["ReAct: Out-of-distribution Detection With Rectified Activations","Yiyou Sun and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.12797",0,"","benchmarks robustness"],["[linkpost] Acquisition of Chess Knowledge in AlphaZero","Quintin Pope","2021","blog","LessWrong","www.lesswrong.com/posts/9NNB9Fc8NTc9RYiFD/linkpost-acquisition-of-chess-knowledge-in-alphazero",0,"","interpretability"],["AI Safety Needs Great Engineers","Andy Jones","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YDF7XhMThhNfHfim9/ai-safety-needs-great-engineers",0,"","robustness"],["AI Safety researcher career review","Benjamin_Todd","2021","blog","EA Forum","forum.effectivealtruism.org/posts/KHw3ezJzA7z3itWNW/ai-safety-researcher-career-review",0,"",""],["AI Tracker: monitoring current and near-future risks from superscale models","Edouard Harris and Jeremie Harris","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/eELny7y7JtCs6fQaA/ai-tracker-monitoring-current-and-near-future-risks-from",0,"","governance forecasting monitoring"],["Integrating Three Models of (Human) Cognition","jbkjr","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6chtMKXpLcJ26t7n5/integrating-three-models-of-human-cognition",0,"","agents"],["Minimal-trust investigations","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/minimal-trust-investigations/",0,"",""],["Slightly advanced decision theory 102: Four reasons not to be a (naive) utility maximizer","Jan","2021","blog","LessWrong","www.lesswrong.com/posts/xGPXDNGYebD3rgCoa/slightly-advanced-decision-theory-102-four-reasons-not-to-be",0,"","theory"],["What is most confusing to you about AI stuff?","Sam Clarke","2021","blog","EA Forum","forum.effectivealtruism.org/posts/viuH9MnnktKF8sY3H/what-is-most-confusing-to-you-about-ai-stuff",0,"",""],["Branching Time Active Inference: empirical study and complexity class analysis","Théophile Champion and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.11276",0,"","agents robustness"],["Morally underdefined situations can be deadly","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7q6jQ7y9xQNAkbTtt/morally-underdefined-situations-can-be-deadly",0,"",""],["Potential Alignment mental tool: Keeping track of the types","Donald Hobson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QEHb8tWLztMyvrv6f/potential-alignment-mental-tool-keeping-track-of-the-types",0,"",""],["Some real examples of gradient hacking","Oliver Sourbut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GPoPKk2wr2r4uPwxe/some-real-examples-of-gradient-hacking",0,"",""],["Yudkowsky and Christiano discuss \"Takeoff Speeds\"","Eliezer Yudkowsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vwLxd6hhFvPbvKmBH/yudkowsky-and-christiano-discuss-takeoff-speeds",0,"","deception forecasting robustness"],["Yudkowsky and Christiano discuss \"Takeoff Speeds\"","EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/rho5vtxSaEdXxLu3o/yudkowsky-and-christiano-discuss-takeoff-speeds",0,"","governance forecasting"],["Yudkowsky and Christiano discuss “Takeoff Speeds”","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/22/yudkowsky-and-christiano-discuss-takeoff-speeds/",0,"","forecasting"],["From language to ethics by automated reasoning","Michele Campolo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/uyvnjaRaKdGXoKrv7/from-language-to-ethics-by-automated-reasoning",0,"",""],["Genuineness, Existential Selfdetermination, Satisfaction: pick 2","Tamsin Leake","2021","blog","carado.moe","carado.moe/genuineness-existselfdet-satisfaction-pick2.html",0,"",""],["the two-vtable problem","Tamsin Leake","2021","blog","carado.moe","carado.moe/two-vtable.html",0,"",""],["A Certain Formalization of Corrigibility Is VNM-Incoherent","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/WCX3EwnWAx7eyucqH/a-certain-formalization-of-corrigibility-is-vnm-incoherent",0,"","instrumental-convergence"],["Discrete Representations Strengthen Vision Transformer Robustness","Chengzhi Mao and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.10493",0,"","benchmarks robustness"],["endiannesses","Tamsin Leake","2021","blog","carado.moe","carado.moe/endiannesses.html",0,"",""],["More detailed proposal for measuring alignment of current models","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GbAymLbJdGbqTumCN/more-detailed-proposal-for-measuring-alignment-of-current",0,"",""],["no room above paperclips","Tamsin Leake","2021","blog","carado.moe","carado.moe/above-paperclips.html",0,"",""],["rust & wasm, without wasm-pack","Tamsin Leake","2021","blog","carado.moe","carado.moe/rust-wasm-without-wasmpack.html",0,"",""],["unoptimal superintelligence loses","Tamsin Leake","2021","blog","carado.moe","carado.moe/unoptimal-superint-loses.html",0,"",""],["Goodhart: Endgame","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dmp9PZjpSSX5NeXHM/goodhart-endgame",0,"","goodharts-law"],["How To Get Into Independent Research On Alignment/Agency","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/P3Yt66Wh5g7SbkKuT/how-to-get-into-independent-research-on-alignment-agency",0,"",""],["Ngo and Yudkowsky on AI capability gains","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/18/ngo-and-yudkowsky-on-ai-capability-gains/",0,"",""],["Ngo and Yudkowsky on AI capability gains","richard_ngo and EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/v2KL4ApqrxuYqQckK/ngo-and-yudkowsky-on-ai-capability-gains",0,"","governance"],["Tool-assisted speedrunning","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/tool-assisted-speedrunning/",0,"",""],["Finding Useful Predictions by Meta-gradient Descent to Improve Decision-making","Alex Kearney and 3 others","2021","paper","NeurIPS 2021 Workshop on Self-Supervised Learning: Theory and\n  Practice","arxiv.org/abs/2111.11212",0,"","agents"],["Ngo and Yudkowsky on AI capability gains","Eliezer Yudkowsky and Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hwxj4gieR7FWNwYfa/ngo-and-yudkowsky-on-ai-capability-gains-1",0,"","governance forecasting"],["Satisficers Tend To Seek Power: Instrumental Convergence Via Retargetability","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nZY8Np759HYFawdjH/satisficers-tend-to-seek-power-instrumental-convergence-via",0,"","instrumental-convergence"],["Software Engineering for Responsible AI: An Empirical Study and Operationalised Patterns","Qinghua Lu and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.09478",0,"",""],["“Biological anchors” is about bounding, not pinpointing, AI timelines","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/biological-anchors-is-about-bounding-not-pinpointing-ai-timelines/",0,"","forecasting"],["Acquisition of Chess Knowledge in AlphaZero","Thomas McGrath and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.09259",0,"","interpretability agents"],["Applications for AI Safety Camp 2022 Now Open!","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QeetPm8yvFf7mAGj9/applications-for-ai-safety-camp-2022-now-open",0,"",""],["A positive case for how we might succeed at prosaic AI alignment","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5ciYedyQDDqAcrDLr/a-positive-case-for-how-we-might-succeed-at-prosaic-ai",0,"",""],["Artificial Intelligence Needs Environmental Ethics | Global Catastrophic Risk Institute","Seth Baum","2021","report","gcrinstitute.org","gcrinstitute.org/artificial-intelligence-needs-environmental-ethics/",0,"",""],["Falling everyday violence, bigger wars and atrocities: how do they net out?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/has-violence-declined-when-we-include-the-world-wars-and-other-major-atrocities/",0,"",""],["Improving Learning from Demonstrations by Learning from Experience","Haofeng Liu and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.08156",0,"","agents robustness"],["Ngo and Yudkowsky on alignment difficulty","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/15/ngo-and-yudkowsky-on-alignment-difficulty/",0,"",""],["Quantilizer ≡ Optimizer with a Bounded Amount of Output","itaibn0","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZjDh3BmbDrWJRckEb/quantilizer-optimizer-with-a-bounded-amount-of-output-1",0,"",""],["Solving Probability and Statistics Problems by Program Synthesis","Leonard Tang and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.08267",0,"",""],["Two Stupid AI Alignment Ideas","aphyer","2021","blog","LessWrong","www.lesswrong.com/posts/AtzuxdKs9DXcD7G6o/two-stupid-ai-alignment-ideas",0,"",""],["Attempted Gears Analysis of AGI Intervention Discussion With Eliezer","Zvi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xHnuX42WNZ9hq53bz/attempted-gears-analysis-of-agi-intervention-discussion-with-1",0,"",""],["My understanding of the alignment problem","danieldewey","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/WckCfXpfrb9Bms8aA/my-understanding-of-the-alignment-problem",0,"",""],["Ngo and Yudkowsky on alignment difficulty","Eliezer Yudkowsky and Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7im8at9PmhbT4JHsW/ngo-and-yudkowsky-on-alignment-difficulty",0,"","deception power-seeking automated-alignment-research forecasting"],["Ngo and Yudkowsky on alignment difficulty","richard_ngo and EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/iGYTt3qvJFGppxJbk/ngo-and-yudkowsky-on-alignment-difficulty",0,"",""],["\"Slower tech development\" can be about ordering, gradualness, or distance from now","MichaelA","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ujRGGBxJN9AHXfzJe/slower-tech-development-can-be-about-ordering-gradualness-or",0,"","forecasting"],["Artificial Intelligence Needs Environmental Ethics","Andrea Owe and Seth Baum","2021","report","gcrinstitute.org","gcrinstitute.org/papers/059_ai-environmental-ethics.pdf",0,"",""],["What would we do if alignment were futile?","Grant Demaree","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xv77XjuZEkjRsvkJp/what-would-we-do-if-alignment-were-futile",0,"",""],["A FLI postdoctoral grant application: AI alignment via causal analysis and design of agents","PabloAMC","2021","blog","LessWrong","www.lesswrong.com/posts/9md9QtHmhNnAmacdu/a-fli-postdoctoral-grant-application-ai-alignment-via-causal",0,"","agents"],["Comments on Carlsmith's “Is power-seeking AI an existential risk?”","So8res","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cCMihiwtZx7kdcKgt/comments-on-carlsmith-s-is-power-seeking-ai-an-existential",0,"","power-seeking forecasting"],["Is Functional Decision Theory still an active area of research?","Grant Demaree","2021","blog","LessWrong","www.lesswrong.com/posts/Ertfzigmjwx3YPGqG/is-functional-decision-theory-still-an-active-area-of",0,"","theory"],["What’s the likelihood of only sub exponential growth for AGI?","M. Y. Zuo","2021","blog","LessWrong","www.lesswrong.com/posts/3H8bmvgqBBpk48Dgn/what-s-the-likelihood-of-only-sub-exponential-growth-for-agi",0,"","forecasting robustness"],["A Defense of Functional Decision Theory","Heighn","2021","blog","LessWrong","www.lesswrong.com/posts/R8muGSShCXZEnuEi6/a-defense-of-functional-decision-theory",0,"","theory"],["Human irrationality: both bad and good for reward inference","Lawrence Chan and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.06956",0,"","robustness"],["What would you do if you had a lot of money/power/influence and you thought that AI timelines were very short?","Greg_Colbourn","2021","blog","EA Forum","forum.effectivealtruism.org/posts/wrdWS2K8hWfoAzRst/what-would-you-do-if-you-had-a-lot-of-money-power-influence",0,"","governance forecasting"],["Why I'm excited about Redwood Research's current project","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pXLqpguHJzxSjDdx7/why-i-m-excited-about-redwood-research-s-current-project",0,"","robustness"],["Discussion with Eliezer Yudkowsky on AGI interventions","Rob Bensinger and Eliezer Yudkowsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CpvyhFy9WvCNsifkY/discussion-with-eliezer-yudkowsky-on-agi-interventions",0,"",""],["Discussion with Eliezer Yudkowsky on AGI interventions","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/11/discussion-with-eliezer-yudkowsky-on-agi-interventions/",0,"",""],["Discussion with Eliezer Yudkowsky on AGI interventions","RobBensinger and EliezerYudkowsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/bGBm2yTiLEwwCbL6w/discussion-with-eliezer-yudkowsky-on-agi-interventions",0,"","governance"],["Explainable AI (XAI): A Systematic Meta-Survey of Current Challenges and Future Opportunities","Waddah Saeed and Christian Omlin","2021","paper","arXiv preprint","arxiv.org/abs/2111.06420",0,"","interpretability"],["Model-Free Risk-Sensitive Reinforcement Learning","DeepMind Safety Research","2021","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/model-free-risk-sensitive-reinforcement-learning-5a12ba5ce662",0,"",""],["Reflections on the first year of parenting","Victoria Krakovna","2021","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2021/11/11/reflections-on-the-first-year-of-parenting/",0,"",""],["Weak point in “most important century”: lock-in","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/weak-point-in-most-important-century-lock-in/",0,"",""],["BERI is hiring an ML Software Engineer","sawyer","2021","blog","EA Forum","forum.effectivealtruism.org/posts/Nn2eudXZsRHx2xvti/beri-is-hiring-an-ml-software-engineer",0,"",""],["What exactly is GPT-3's base objective?","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Nq58w4SiZMjHdAPaX/what-exactly-is-gpt-3-s-base-objective",0,"",""],["Building an AI-ready RSE Workforce","Ying Zhang and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.04916",0,"",""],["Data Augmentation Can Improve Robustness","Sylvestre-Alvise Rebuffi and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.05328",0,"","evals robustness"],["Long-term AI policy strategy research and implementation","Benjamin_Todd","2021","blog","EA Forum","forum.effectivealtruism.org/posts/da4S6pbkbQ8azcfSA/long-term-ai-policy-strategy-research-and-implementation",0,"","governance policy"],["Lymph Node Detection in T2 MRI with Transformers","Tejas Sudharshan Mathai and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.04885",0,"",""],["Possible research directions to improve the mechanistic explanation of neural networks","delton137","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HiutLvY2x7zrsTQkx/possible-research-directions-to-improve-the-mechanistic",0,"","interpretability"],["Rowing, Steering, Anchoring, Equity, Mutiny","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/rowing-steering-anchoring-equity-mutiny/",0,"",""],["Unsupervised deep learning identifies semantic disentanglement in single inferotemporal face patch neurons","Irina Higgins and 6 others","2021","blog","deepmind.com","www.deepmind.com/blog/unsupervised-deep-learning-identifies-semantic-disentanglement-in-single-inferotemporal-face-patch-neurons",0,"",""],["against AI alignment ?","Tamsin Leake","2021","blog","carado.moe","carado.moe/against-ai-alignment.html",0,"",""],["Efficient estimates of optimal transport via low-dimensional embeddings","Patric M. Fulop and Vincent Danos","2021","paper","arXiv preprint","arxiv.org/abs/2111.04838",0,"",""],["How do we become confident in the safety of a machine learning system?","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FDJnZt8Ks2djouQTZ/how-do-we-become-confident-in-the-safety-of-a-machine",0,"",""],["psi: a universal format for structured information","Tamsin Leake","2021","blog","carado.moe","carado.moe/psi.html",0,"",""],["What are red flags for Neural Network suffering?","Marius Hobbhahn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Bpw2HXjMa3GaouDnC/what-are-red-flags-for-neural-network-suffering",0,"",""],["A Word on Machine Ethics: A Response to Jiang et al. (2021)","Zeerak Talat and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.04158",0,"","interpretability"],["Using Brain-Computer Interfaces to get more data for AI alignment","Robbo","2021","blog","LessWrong","www.lesswrong.com/posts/iWv6Pu2fWPKqevzFE/using-brain-computer-interfaces-to-get-more-data-for-ai",0,"",""],["[Discussion] Best intuition pumps for AI safety","mariushobbhahn","2021","blog","EA Forum","forum.effectivealtruism.org/posts/cFRbLmhCEu74wcJ3D/discussion-best-intuition-pumps-for-ai-safety",0,"",""],["Chu are you?","Adele Lopez","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/89EvBkc4nkbEctzR3/chu-are-you",0,"",""],["Linguistic Cues of Deception in a Multilingual April Fools' Day Context","Katerina Papantoniou and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.03913",0,"","deception"],["November 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/11/06/november-2021-newsletter/",0,"",""],["Comments on OpenPhil's Interpretability RFP","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oWN9fgYnFYJEWdAs9/comments-on-openphil-s-interpretability-rfp",0,"","interpretability"],["Drug addicts and deceptively aligned agents - a comparative analysis","Jan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pFXEG9C5m2X5h2yiq/drug-addicts-and-deceptively-aligned-agents-a-comparative",0,"","deception agents"],["Modeling the impact of safety agendas","Ben Cottier","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/t7f6gF2kpafCMw6rv/modeling-the-impact-of-safety-agendas",0,"",""],["Adversarial GLUE: A Multi-Task Benchmark for Robustness Evaluation of Language Models","Boxin Wang and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.02840",0,"","evals benchmarks robustness"],["B-Pref: Benchmarking Preference-Based Reinforcement Learning","Kimin Lee and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.03026",0,"","evals benchmarks deception robustness"],["Hunter-gatherer happiness","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/hunter-gatherer-happiness/",0,"",""],["Apply to the ML for Alignment Bootcamp (MLAB) in Berkeley [Jan 3 - Jan 22]","habryka and Buck","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YgpDYjTx7DCEgziG5/apply-to-the-ml-for-alignment-bootcamp-mlab-in-berkeley-jan",0,"",""],["Apply to the ML for Alignment Bootcamp (MLAB) in Berkeley [Jan 3 - Jan 22]","Habryka and Buck","2021","blog","EA Forum","forum.effectivealtruism.org/posts/iwTr8S8QkutyYroGy/apply-to-the-ml-for-alignment-bootcamp-mlab-in-berkeley-jan",0,"","robustness"],["How to Improve China-Western Coordination on EA Issues?","Michael Kehoe","2021","blog","EA Forum","forum.effectivealtruism.org/posts/NhnpD6Pt4ZZtouQre/how-to-improve-china-western-coordination-on-ea-issues",0,"","governance policy"],["The case for long-term corporate governance of AI","SethBaum and jonasschuett","2021","blog","EA Forum","forum.effectivealtruism.org/posts/5MZpxbJJ5pkEBpAAR/the-case-for-long-term-corporate-governance-of-ai",0,"","governance"],["AI Ethics Statements -- Analysis and lessons learnt from NeurIPS Broader Impact Statements","Carolyn Ashurst and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.01705",0,"","interpretability governance"],["EfficientZero: human ALE sample-efficiency w/MuZero+self-supervised","gwern","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jYNT3Qihn2aAYaaPb/efficientzero-human-ale-sample-efficiency-w-muzero-self",0,"",""],["Models Modeling Models","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nA3n2vfCy3ffnjapw/models-modeling-models",0,"","goodharts-law"],["Unraveling the evidence about violence among very early humans","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/unraveling-the-evidence-about-violence-among-very-early-humans/",0,"",""],["Artificial intelligence, systemic risks, and sustainability","Victor Galaz and 14 others","2021","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0160791X21002165",0,"",""],["Classifying AI Systems","Catherine Aiken","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/classifying-ai-systems/",0,"",""],["Feature Selection","Zack_M_Davis","2021","blog","LessWrong","www.lesswrong.com/posts/dYspinGtiba5oDCcv/feature-selection",0,"",""],["Federal Prize Competitions","Ali Crawford and Ido Wulkan","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/federal-prize-competitions/",0,"",""],["Moral consideration of nonhumans in the ethics of artificial intelligence","Andrea Owe and Seth D. Baum","2021","report","link.springer.com","link.springer.com/10.1007/s43681-021-00065-0",0,"",""],["saving the server-side of the internet: just WASM,","Tamsin Leake","2021","blog","carado.moe","carado.moe/saving-server-internet.html",0,"",""],["Apply to be a Stanford HAI Junior Fellow (Assistant Professor- Research) by Nov. 15, 2021","Vael Gates","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ed9EHSDLRp2oMwoyr/apply-to-be-a-stanford-hai-junior-fellow-assistant-professor",0,"",""],["Nate Soares on the Ultimate Newcomb's Problem","Rob Bensinger","2021","blog","LessWrong","www.lesswrong.com/posts/F3aESx4JWWEDAFEiH/nate-soares-on-the-ultimate-newcomb-s-problem",0,"","theory"],["A very crude deception eval is already passed","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/22GrdspteQc8EonMn/a-very-crude-deception-eval-is-already-passed",0,"","evals deception"],["Cold Links: nonfiction yarns","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-nonfiction-yarns-2/",0,"",""],["Interpretability","abergal and Nick_Beckstead","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CzZ6Fch4JSpwCpu6C/interpretability",0,"","interpretability"],["Learning to Be Cautious","Montaser Mohammedalamen and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.15907",0,"","agents policy"],["Measuring and forecasting risks","abergal and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7DhwRLoKm4nMrFFsH/measuring-and-forecasting-risks",0,"","forecasting"],["Request for proposals for projects in AI alignment that work with deep learning systems","abergal and Nick_Beckstead","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/H5iePjNKaaYQyZpgR/request-for-proposals-for-projects-in-ai-alignment-that-work",0,"",""],["Stuart Russell and Melanie Mitchell on Munk Debates","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GDnFsyfKedevKHAuJ/stuart-russell-and-melanie-mitchell-on-munk-debates",0,"",""],["Techniques for enhancing human feedback","abergal and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ybThg9nA7u6f8qfZZ/techniques-for-enhancing-human-feedback",0,"","rlhf"],["Truthful and honest AI","abergal and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/sdxZdGFtAwHGFGKhg/truthful-and-honest-ai",0,"",""],["[AN #168]: Four technical topics for which Open Phil is soliciting grant proposals","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AXj9KSvda6XwNwLrS/an-168-four-technical-topics-for-which-open-phil-is",0,"",""],["Forecasting progress in language models","Matthew Barnett and Metaculus","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tepqESMuRmyhtmDS7/forecasting-progress-in-language-models",0,"","forecasting"],["Selfishness, preference falsification, and AI alignment","jessicata","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZddY8BZbvoXHEvDHf/selfishness-preference-falsification-and-ai-alignment",0,"",""],["Weak point in “most important century”: full automation","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/weak-point-in-most-important-century-full-automation/",0,"",""],["Toward a Theory of Justice for Artificial Intelligence","Iason Gabriel","2021","paper","arXiv preprint","arxiv.org/abs/2110.14419",0,"","robustness"],["Understanding Interlocking Dynamics of Cooperative Rationalization","Mo Yu and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.13880",0,"","benchmarks policy"],["Was life better in hunter-gatherer times?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/was-life-better-in-hunter-gatherer-times/",0,"",""],["A Preliminary Exploration into Factored Cognition with Language Models","Leo Gao and 3 others","2021","blog","blog.eleuther.ai","blog.eleuther.ai/factored-cognition/",0,"",""],["QuantifyML: How Good is my Machine Learning Model?","Muhammad Usman and 2 others","2021","paper","EPTCS 348, 2021, pp. 92-100","arxiv.org/abs/2110.12588",0,"","evals robustness"],["What Would Jiminy Cricket Do? Towards Agents That Behave Morally","Dan Hendrycks and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.13136",0,"","evals agents"],["lamenting nerds","Tamsin Leake","2021","blog","carado.moe","carado.moe/lamenting-nerds.html",0,"",""],["Phil Trammell on Economic Growth Under Transformative AI","Michaël Trazzi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BNoHokwCiPGmFHnp8/phil-trammell-on-economic-growth-under-transformative-ai",0,"",""],["P₂B: Plan to P₂B Better","Ramana Kumar and Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CAwwFpbteYBQw2Gkp/p-b-plan-to-p-b-better",0,"","instrumental-convergence"],["Towards Deconfusing Gradient Hacking","leogao","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/u3fP8vjGsDCT7X54H/towards-deconfusing-gradient-hacking",0,"",""],["Inference cost limits the impact of ever larger models","SoerenMind","2021","blog","LessWrong","www.lesswrong.com/posts/zTDkhm6yFq6edhZ7L/inference-cost-limits-the-impact-of-ever-larger-models",0,"","forecasting"],["alignment is an optimization processes problem","Tamsin Leake","2021","blog","carado.moe","carado.moe/alignment-optimization-processes.html",0,"",""],["AMA on Truthful AI: Owen Cotton-Barratt, Owain Evans & co-authors","Owain_Evans","2021","blog","LessWrong","www.lesswrong.com/posts/mwTEMHKv9tG9HxFXD/ama-on-truthful-ai-owen-cotton-barratt-owain-evans-and-co",0,"","governance"],["Epistemic Strategies of Safety-Capabilities Tradeoffs","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hWag6E7XPCbdfaoKZ/epistemic-strategies-of-safety-capabilities-tradeoffs",0,"",""],["General alignment plus human values, or alignment via human values?","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3e6pmovj6EJ729M2i/general-alignment-plus-human-values-or-alignment-via-human",0,"",""],["Emergent modularity and safety","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zvEbeZ6opjPJiQnFE/emergent-modularity-and-safety",0,"",""],["Ethical Norms for New Generation Artificial Intelligence Released","PRC Ministry of Science and Technology","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ethical-norms-for-new-generation-artificial-intelligence-released/",0,"",""],["Podcast: Krister Bykvist on moral uncertainty, rationality, metaethics, AI and future populations","Gus Docker","2021","blog","EA Forum","forum.effectivealtruism.org/posts/oeAdt2GukZ3KayFhM/podcast-krister-bykvist-on-moral-uncertainty-rationality",0,"",""],["to wasm and back again: the essence of portable programs","Tamsin Leake","2021","blog","carado.moe","carado.moe/portable-programs.html",0,"",""],["[AN #167]: Concrete ML safety problems and their relevance to x-risk","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/suy5w8cWZJZsv2XES/an-167-concrete-ml-safety-problems-and-their-relevance-to-x",0,"",""],["AGI Safety Fundamentals curriculum and application","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zmwkz2BMvuFFR8bi3/agi-safety-fundamentals-curriculum-and-application",0,"",""],["AGI Safety Fundamentals curriculum and application","richard_ngo","2021","blog","EA Forum","forum.effectivealtruism.org/posts/BpAKCeGMtQqqty9ZJ/agi-safety-fundamentals-curriculum-and-application",0,"","governance"],["Reading books vs. engaging with them","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/reading-books-vs-engaging-with-them/",0,"",""],["Shaking the foundations: delusions in sequence models for interaction and control","Pedro A. Ortega and 18 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.10819",0,"",""],["Truthful AI","Owen Cotton-Barratt and 2 others","2021","blog","EA Forum","forum.effectivealtruism.org/posts/SGFRneArKi93qbrRG/truthful-ai",0,"","governance"],["Pre-agriculture gender relations seem bad","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/hunter-gatherer-gender-relations-seem-bad/",0,"",""],["Risks of AI Foundation Models in Education","Su Lin Blodgett and Michael Madaio","2021","paper","arXiv preprint","arxiv.org/abs/2110.10024",0,"",""],["[Creative Writing Contest] An AI Safety Limerick","Ben_West","2021","blog","EA Forum","forum.effectivealtruism.org/posts/udGrjhfYqxv7GhWA4/creative-writing-contest-an-ai-safety-limerick",0,"",""],["[MLSN #1]: ICLR Safety Paper Roundup","Dan H","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8Gv5zSCnGeLxK5FAF/mlsn-1-iclr-safety-paper-roundup",0,"",""],["An ML safety insurance company - shower thoughts","EdoArad","2021","blog","EA Forum","forum.effectivealtruism.org/posts/D6pbrzLcMaMsQNmNb/an-ml-safety-insurance-company-shower-thoughts",0,"",""],["Beyond the human training distribution: would the AI CEO create almost-illegal teddies?","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tHChCJB9piCTD7HEx/beyond-the-human-training-distribution-would-the-ai-ceo",0,"",""],["Epistemic Strategies of Selection Theorems","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LWmmfTvptiJp7wvFg/epistemic-strategies-of-selection-theorems",0,"",""],["MEMO: Test Time Robustness via Adaptation and Augmentation","Marvin Zhang and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.09506",0,"","evals benchmarks robustness"],["ML Safety Newsletter #1","Dan Hendrycks","2021","blog","newsletter.mlsafety.org","newsletter.mlsafety.org/p/ml-safety-newsletter-1",0,"",""],["New Working Paper Series of the Legal Priorities Project","Legal Priorities Project","2021","blog","EA Forum","forum.effectivealtruism.org/posts/uXNytB4fhgSyMJp9w/new-working-paper-series-of-the-legal-priorities-project",0,"","governance policy"],["On The Risks of Emergent Behavior in Foundation Models","jsteinhardt","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QmdrkuArFHphqANRE/on-the-risks-of-emergent-behavior-in-foundation-models",0,"",""],["Truthful AI: Developing and governing AI that does not lie","Owain_Evans and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/aBixCPqSnTsPsTJBQ/truthful-ai-developing-and-governing-ai-that-does-not-lie",0,"","governance"],["Value alignment: a formal approach","Carles Sierra and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.09240",0,"","agents"],["Improving End-To-End Modeling for Mispronunciation Detection with Effective Augmentation Mechanisms","Tien-Hong Lo and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.08731",0,"","training-data"],["[Creative Writing Contest] Metal or Mortal","Louis","2021","blog","EA Forum","forum.effectivealtruism.org/posts/94DvC7J5vtbSKWSXf/creative-writing-contest-metal-or-mortal",0,"",""],["Analyzing Dynamic Adversarial Training Data in the Limit","Eric Wallace and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.08514",0,"","robustness training-data"],["Memetic hazards of AGI architecture posts","Ozyrus","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/XTGceK7xE8rxrRLcr/memetic-hazards-of-agi-architecture-posts-1",0,"",""],["Optimization Concepts in the Game of Life","Vika and Ramana Kumar","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/mL8KdftNGBScmBcBg/optimization-concepts-in-the-game-of-life",0,"","theory"],["The AGI needs to be honest","rokosbasilisk","2021","blog","LessWrong","www.lesswrong.com/posts/hqzHbew35Jx4xoDhE/the-agi-needs-to-be-honest",0,"",""],["Cold Links: assorted sports longreads","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-assorted-sports-longreads/",0,"",""],["Collaborating with Humans without Human Data","DJ Strouse and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.08176",0,"","agents"],["Evaluating the Faithfulness of Importance Measures in NLP by Recursively Masking Allegedly Important Tokens and Retraining","Andreas Madsen and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.08412",0,"","evals"],["General vs specific arguments for the longtermist importance of shaping AI development","Sam Clarke","2021","blog","EA Forum","forum.effectivealtruism.org/posts/n5vRgmv3iBEe6Xh3P/general-vs-specific-arguments-for-the-longtermist-importance",0,"",""],["NLP Position Paper: When Combatting Hype, Proceed with Caution","Sam Bowman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RLHkSBQ7zmTzAjsio/nlp-position-paper-when-combatting-hype-proceed-with-caution",0,"",""],["Robustness of different loss functions and their impact on networks learning capability","Vishal Rajput","2021","paper","arXiv preprint","arxiv.org/abs/2110.08322",0,"","agents robustness"],["Can Machines Learn Morality? The Delphi Experiment","Liwei Jiang and 14 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.07574",0,"",""],["Classical symbol grounding and causal graphs","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SnKfFscgC8Nj5ddi3/classical-symbol-grounding-and-causal-graphs",0,"",""],["Compute Governance and Conclusions - Transformative AI and Compute [3/4]","lennart","2021","blog","LessWrong","www.lesswrong.com/posts/M3xpp7CZ2JaSafDJB/compute-governance-and-conclusions-transformative-ai-and",0,"","governance compute-governance forecasting scaling-laws"],["If I were a billion years old","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/olden-the-imaginary-billion-year-old-version-of-me/",0,"",""],["[Creative Writing Contest] The Puppy Problem","Louis","2021","blog","EA Forum","forum.effectivealtruism.org/posts/MwJiR6WKgPsTiADSK/creative-writing-contest-the-puppy-problem",0,"",""],["[Proposal] Method of locating useful subnets in large models","Quintin Pope","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ji6hQbwH7tK7mejhk/proposal-method-of-locating-useful-subnets-in-large-models",0,"","interpretability"],["AI Vignettes Project","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/ai-vignettes-project/",0,"",""],["cosmic missing outs","Tamsin Leake","2021","blog","carado.moe","carado.moe/cosmic-missing-outs.html",0,"",""],["Is it crunch time yet? If so, who can help?","NicholasKross","2021","blog","EA Forum","forum.effectivealtruism.org/posts/3jfXxzxrnwPwBwiig/is-it-crunch-time-yet-if-so-who-can-help",0,"",""],["New new blog location","jsteinhardt","2021","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2021/10/13/new-new-blog-location/",0,"",""],["Quantifying Local Specialization in Deep Neural Networks","Shlomi Hod and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.08058",0,"","robustness"],["AI Risk in Africa","Claude Formanek","2021","blog","EA Forum","forum.effectivealtruism.org/posts/wLQkTBHcPKtoku8Js/ai-risk-in-africa",0,"","governance"],["EDT with updating double counts","paulfchristiano","2021","blog","LessWrong","www.lesswrong.com/posts/m3DiiBiXApN3kQMyM/edt-with-updating-double-counts",0,"","theory"],["exact minds in an exact world","Tamsin Leake","2021","blog","carado.moe","carado.moe/exact-minds-in-an-exact-world.html",0,"",""],["Has life gotten better?: the post-industrial era","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/has-life-gotten-better-the-post-industrial-era/",0,"",""],["Modeling Risks From Learned Optimization","Ben Cottier","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/T9oFjteStcE2ijCJi/modeling-risks-from-learned-optimization",0,"",""],["Certified Patch Robustness via Smoothed Vision Transformers","Hadi Salman and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.07719",0,"","robustness"],["Multiple Choice Normalization in LM Evaluation","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/multiple-choice-normalization/",0,"","evals"],["NVIDIA and Microsoft releases 530B parameter transformer model, Megatron-Turing NLG","Ozyrus","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bGuMrzhJdENCo8BxX/nvidia-and-microsoft-releases-530b-parameter-transformer",0,"","scaling-laws"],["On Solving Problems Before They Appear: The Weird Epistemologies of Alignment","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FQqcejhNWGG8vHDch/on-solving-problems-before-they-appear-the-weird",0,"",""],["meta-tracking","Tamsin Leake","2021","blog","carado.moe","carado.moe/metatracking.html",0,"",""],["The evaluation function of an AI is not its aim","Yair Halberstadt","2021","blog","LessWrong","www.lesswrong.com/posts/trA3wEA7oXw3TF4ho/the-evaluation-function-of-an-ai-is-not-its-aim",0,"","evals"],["The Extrapolation Problem","lsusr","2021","blog","LessWrong","www.lesswrong.com/posts/ASxdfSKTbcEy6MCr3/the-extrapolation-problem",0,"","forecasting"],["Why aren't you freaking out about OpenAI? At what point would you start?","AppliedDivinityStudies","2021","blog","EA Forum","forum.effectivealtruism.org/posts/fmDFytmxwX9qBgcaX/why-aren-t-you-freaking-out-about-openai-at-what-point-would",0,"",""],["Intelligence or Evolution?","Ramana Kumar","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/TATWqHvxKEpL34yKz/intelligence-or-evolution",0,"",""],["Steelman arguments against the idea that AGI is inevitable and will arrive soon","RomanS","2021","blog","LessWrong","www.lesswrong.com/posts/3kijTbgfizDgSgst3/steelman-arguments-against-the-idea-that-agi-is-inevitable",0,"","forecasting"],["[AN #166]: Is it crazy to claim we're in the most important century?","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YbAZ4nSA8itL2EkDb/an-166-is-it-crazy-to-claim-we-re-in-the-most-important",0,"",""],["October 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/10/07/october-2021-newsletter/",0,"",""],["Fingerprinting Multi-exit Deep Neural Network Models via Inference Time","Tian Dong and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.03175",0,"","robustness"],["Safety-capabilities tradeoff dials are inevitable in AGI","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tmyTb4bQQi7C47sde/safety-capabilities-tradeoff-dials-are-inevitable-in-agi",0,"",""],["“Technological unemployment” AI vs. “most important century” AI: how far apart?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/technological-unemployment-ai-vs-most-important-century-ai-how-far-apart/",0,"",""],["[Job ad] Research important longtermist topics at Rethink Priorities!","Linch","2021","blog","EA Forum","forum.effectivealtruism.org/posts/3vXXthjBKhNo8sgFv/job-ad-research-important-longtermist-topics-at-rethink",0,"","governance"],["Automated Fact Checking: A Look at the Field","Hoagy","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CWD8FxA3yJPmZE9o3/automated-fact-checking-a-look-at-the-field",0,"",""],["Preferences from (real and hypothetical) psychology papers","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FsxPNRJ5NQkrSKyDx/preferences-from-real-and-hypothetical-psychology-papers",0,"",""],["We're Redwood Research, we do applied alignment research, AMA","Nate Thomas","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qHvHvBR8L6oycnMXe/we-re-redwood-research-we-do-applied-alignment-research-ama",0,"","robustness"],["Force neural nets to use models, then detect these","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pdJqEzbQrucTEF6DW/force-neural-nets-to-use-models-then-detect-these",0,"",""],["Has Life Gotten Better?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/has-life-gotten-better/",0,"",""],["Procedure Planning in Instructional Videos via Contextual Modeling and Model-based Policy Learning","Jing Bi and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.01770",0,"","policy"],["Thinking Fast and Slow in AI: the Role of Metacognition","Marianna Bergamaschi Ganapini and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.01834",0,"","agents"],["Learning to Assist Agents by Observing Them","Antti Keurulainen and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.01311",0,"","agents policy"],["Nuclear Espionage and AI Governance","GAA","2021","blog","EA Forum","forum.effectivealtruism.org/posts/CKfHDw5Lmoo6jahZD/nuclear-espionage-and-ai-governance-1",0,"","governance"],["Nuclear Espionage and AI Governance","GAA","2021","blog","LessWrong","www.lesswrong.com/posts/GwxotzGc2ipRNweg3/nuclear-espionage-and-ai-governance",0,"","governance"],["A Framework of Prediction Technologies","isaduan","2021","blog","LessWrong","www.lesswrong.com/posts/ZD5meZwgfFJD2wDB5/a-framework-of-prediction-technologies-1",0,"","forecasting"],["Occam's Razor and the Universal Prior","Peter Chatain","2021","blog","LessWrong","www.lesswrong.com/posts/RKfg86eKQuqLnjGxx/occam-s-razor-and-the-universal-prior",0,"","theory"],["The Dark Side of Cognition Hypothesis","Cameron Berg","2021","blog","LessWrong","www.lesswrong.com/posts/8GY7LTFHuitFqwAaH/the-dark-side-of-cognition-hypothesis",0,"",""],["Why does (any particular) AI safety work reduce s-risks more than it increases them?","MichaelStJules","2021","blog","EA Forum","forum.effectivealtruism.org/posts/b5dctgmwBiKhu3BCP/why-does-any-particular-ai-safety-work-reduce-s-risks-more",0,"",""],["A collection of AI Governance-related Podcasts, Newsletters, Blogs, and more","alexlintz","2021","blog","EA Forum","forum.effectivealtruism.org/posts/G3DbHg6qu3tqFGcRW/a-collection-of-ai-governance-related-podcasts-newsletters",0,"","governance"],["Forecasting Compute - Transformative AI and Compute [2/4]","lennart","2021","blog","LessWrong","www.lesswrong.com/posts/sHAaMpdk9FT9XsLvB/forecasting-compute-transformative-ai-and-compute-2-4",0,"","forecasting scaling-laws"],["Gell-Mann Earworms","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/gell-mann-earworms/",0,"",""],["Harnessed Lightning","Ryan Fedasiuk and 2 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/harnessed-lightning/",0,"",""],["Mapping the AI Investment Activities of Top Global Defense Companies","Ngor Luong and 2 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/mapping-the-ai-investment-activities-of-top-global-defense-companies/",0,"",""],["Meta learning to gradient hack","Quintin Pope","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jh6dkqN2wd7fCRfB5/meta-learning-to-gradient-hack",0,"",""],["No Permits, No Fabs: The Importance of Regulatory Reform for Semiconductor Manufacturing","John VerWey","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/no-permits-no-fabs/",0,"",""],["Proposal: Scaling laws for RL generalization","axioman","2021","blog","LessWrong","www.lesswrong.com/posts/65qmEJHDw3vw69tKm/proposal-scaling-laws-for-rl-generalization",0,"","scaling-laws"],["The Simulation Hypothesis Undercuts the SIA/Great Filter Doomsday Argument","Mark Xu and CarlShulman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HF2vpnmgqmHyLGRrA/the-simulation-hypothesis-undercuts-the-sia-great-filter",0,"",""],["U.S. AI Workforce: Policy Recommendations","Diana Gehlhaus and 3 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/u-s-ai-workforce-policy-recommendations/",0,"","policy"],["What Selection Theorems Do We Expect/Want?","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RuDD3aQWLDSb4eTXP/what-selection-theorems-do-we-expect-want",0,"",""],["AI learns betrayal and how to avoid it","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oeCXS2ZCn4rPyq7LQ/ai-learns-betrayal-and-how-to-avoid-it",0,"",""],["My take on Vanessa Kosoy's take on AGI safety","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SzrmsbkqydpZyPuEh/my-take-on-vanessa-kosoy-s-take-on-agi-safety",0,"",""],["Some Existing Selection Theorems","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/N2NebPD78ioyWHhNm/some-existing-selection-theorems",0,"",""],["Takeoff Speeds and Discontinuities","Sammy Martin and Daniel_Eth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pGXR2ynhe5bBCCNqn/takeoff-speeds-and-discontinuities",0,"","forecasting"],["A brief review of the reasons multi-objective RL could be important in AI Safety Research","Ben Smith","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/i5dLfi6m6FCexReK9/a-brief-review-of-the-reasons-multi-objective-rl-could-be",0,"",""],["Brain-inspired AGI and the \"lifetime anchor\"","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/W6wBmQheDiFmfJqZy/brain-inspired-agi-and-the-lifetime-anchor",0,"","forecasting"],["Self-Supervise, Refine, Repeat: Improving Unsupervised Anomaly Detection","Jinsung Yoon and 5 others","2021","report","openreview.net","openreview.net/forum?id=Nct9j3BVswZ",0,"","monitoring"],["September 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/09/29/september-2021-newsletter/",0,"",""],["Test Time Robustification of Deep Models via Adaptation and Augmentation","Marvin Mengxin Zhang and 2 others","2021","report","openreview.net","openreview.net/forum?id=J1uOGgf-bP",0,"",""],["Unsolved ML Safety Problems","jsteinhardt","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AwMb7C72etphiRvah/unsolved-ml-safety-problems",0,"",""],["Untangling Braids with Multi-agent Q-Learning","Abdullah Khan and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.14502",0,"","agents robustness"],["Argument for AI x-risk from large impacts","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/argument-from-large-impacts/",0,"",""],["Collection of arguments to expect (outer and inner) alignment failure?","Sam Clarke","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/m5or8yzrw9GLavz9b/collection-of-arguments-to-expect-outer-and-inner-alignment",0,"",""],["RAFT: A Real-World Few-Shot Text Classification Benchmark","Neel Alex and 11 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.14076",0,"","evals benchmarks"],["Selection Theorems: A Program For Understanding Agents","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/G2Lne2Fi7Qra5Lbuf/selection-theorems-a-program-for-understanding-agents",0,"","agents robustness"],["Summary of history (empowerment and well-being lens)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/summary-of-history-empowerment-and-well-being-lens/",0,"",""],["[Link post] How plausible are AI Takeover scenarios?","SammyDMartin and Sam Clarke","2021","blog","EA Forum","forum.effectivealtruism.org/posts/KxDgeyyhppRD5qdfZ/link-post-how-plausible-are-ai-takeover-scenarios",0,"","forecasting"],["AI takeoff story: a continuation of progress by other means","Edouard Harris","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fq8ybxtcFvKEsWmF8/ai-takeoff-story-a-continuation-of-progress-by-other-means",0,"","forecasting"],["AISC5 Retrospective: Mechanisms for Avoiding Tragedy of the Commons in Common Pool Resource Problems","Ariel Kwiatkowski and 2 others","2021","blog","LessWrong","www.lesswrong.com/posts/LBwpubeZSi3ottfjs/aisc5-retrospective-mechanisms-for-avoiding-tragedy-of-the",0,"",""],["Beyond fire alarms: freeing the groupstruck","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/beyond-fire-alarms-freeing-the-groupstruck/",0,"","evals"],["The Paradox of Expert Opinion","Emrik","2021","blog","LessWrong","www.lesswrong.com/posts/S6Qcf5EgX5zAozTAa/the-paradox-of-expert-opinion",0,"","goodharts-law"],["Transformative AI and Compute [Summary]","lennart","2021","blog","LessWrong","www.lesswrong.com/posts/XJYdnHQqpengWn3xb/transformative-ai-and-compute-summary-2",0,"","forecasting scaling-laws"],["AXRP Episode 11 - Attainable Utility and Power with Alex Turner","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vcrXS5DmvBuJaKucp/axrp-episode-11-attainable-utility-and-power-with-alex",0,"","instrumental-convergence"],["Cognitive Biases in Large Language Models","Jan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fFF3G4W8FbXigS4gr/cognitive-biases-in-large-language-models",0,"",""],["Pathways: Google's AGI","Lê Nguyên Hoang","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bEKW5gBawZirJXREb/pathways-google-s-agi",0,"",""],["Cartesian Frames and Factored Sets on ArXiv","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MnWQ4o7Y4HryE5ffN/cartesian-frames-and-factored-sets-on-arxiv",0,"",""],["Cold Links: assorted fun basketball stuff","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-assorted-fun-basketball-stuff/",0,"",""],["Fanaticism in AI: SERI Project","Jake Arft-Guatelli","2021","blog","EA Forum","forum.effectivealtruism.org/posts/AC5jfXrBntwgHtZcR/fanaticism-in-ai-seri-project",0,"",""],["Seeking social science students / collaborators interested in AI existential risks","Vael Gates","2021","blog","EA Forum","forum.effectivealtruism.org/posts/dKgWZ8GMNkXfRwjqH/seeking-social-science-students-collaborators-interested-in",0,"",""],["The problem of artificial suffering","mlsbt","2021","blog","EA Forum","forum.effectivealtruism.org/posts/JCBPexSaGCfLtq3DP/the-problem-of-artificial-suffering",0,"",""],["Temporal Inference with Finite Factored Sets","Scott Garrabrant","2021","paper","arXiv preprint","arxiv.org/abs/2109.11513",0,"",""],["The Most Important Century (in a nutshell)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/the-most-important-century-in-a-nutshell/",0,"",""],["What is Compute? - Transformative AI and Compute [1/4]","lennart","2021","blog","LessWrong","www.lesswrong.com/posts/uYXAv6Audr2y4ytJe/what-is-compute-transformative-ai-and-compute-1-4",0,"","forecasting scaling-laws"],["[AN #165]: When large models are more likely to lie","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/NR4rgfKu63TcqLxcH/an-165-when-large-models-are-more-likely-to-lie",0,"",""],["A sufficiently paranoid non-Friendly AGI might self-modify itself to become Friendly","RomanS","2021","blog","LessWrong","www.lesswrong.com/posts/QNCcbW2jLsmw9xwhG/a-sufficiently-paranoid-non-friendly-agi-might-self-modify",0,"","forecasting"],["Cartesian Frames","Scott Garrabrant and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.10996",0,"","agents theory"],["Recursively Summarizing Books with Human Feedback","Jeff Wu and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.10862",0,"","rlhf evals benchmarks"],["UK's new 10-year \"National AI Strategy,\" released today","jared_m","2021","blog","EA Forum","forum.effectivealtruism.org/posts/bBpE5HrjCFDLMvZKd/uk-s-new-10-year-national-ai-strategy-released-today",0,"","governance"],["Announcing the Vitalik Buterin Fellowships in AI Existential Safety!","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hSdgugekxgdyacXTu/announcing-the-vitalik-buterin-fellowships-in-ai-existential",0,"",""],["David Wolpert on Knowledge","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hPPGuiXf3zhKqgCMb/david-wolpert-on-knowledge",0,"",""],["Redwood Research’s current project","Buck","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/k7oxdbNaGATZbtEg3/redwood-research-s-current-project",0,"","policy robustness"],["Why AI alignment could be hard with modern deep learning","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/why-ai-alignment-could-be-hard-with-modern-deep-learning/",0,"","deception sycophancy"],["Why AI alignment could be hard with modern deep learning","Ajeya","2021","blog","EA Forum","forum.effectivealtruism.org/posts/hCsxvMAGpkEuLCE4E/why-ai-alignment-could-be-hard-with-modern-deep-learning",0,"",""],["[Book Review] \"The Alignment Problem\" by Brian Christian","lsusr","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZYDkHWjShKazTywbg/book-review-the-alignment-problem-by-brian-christian",0,"",""],["Actionable Approaches to Promote Ethical AI in Libraries","Helen Bubinger and Jesse David Dinneen","2021","paper","arXiv preprint","arxiv.org/abs/2109.09672",0,"","evals"],["AI, learn to be conservative, then learn to be less so: reducing side-effects, learning preserved features, and going beyond conservatism","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/y2XyxomuEpMaRYDQw/ai-learn-to-be-conservative-then-learn-to-be-less-so",0,"",""],["Is working on AI safety as dangerous as ignoring it?","jkmh","2021","blog","EA Forum","forum.effectivealtruism.org/posts/jSvWKv37DibR8BwNX/is-working-on-ai-safety-as-dangerous-as-ignoring-it",0,"",""],["On Scaling Academia","kirchner.jan","2021","blog","EA Forum","forum.effectivealtruism.org/posts/QbGLmkohgJADdHzsp/on-scaling-academia",0,"",""],["Testing The Natural Abstraction Hypothesis: Project Update","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dNzhdiFE398KcGDc9/testing-the-natural-abstraction-hypothesis-project-update",0,"",""],["What kind of event, targeted to undergraduate CS majors, would be most effective at getting people to work on AI safety?","CBiddulph","2021","blog","EA Forum","forum.effectivealtruism.org/posts/xBDgtTXKfLjggzCn5/what-kind-of-event-targeted-to-undergraduate-cs-majors-would",0,"",""],["Towards Resilient Artificial Intelligence: Survey and Research Issues","Oliver Eigner and 8 others","2021","paper","Proceedings of the 2021 IEEE International Conference on Cyber\n  Security and Resilience (CSR 2021), 2021, 536-542","arxiv.org/abs/2109.08904",0,"",""],["Asimov's Chronology of Science and Discovery","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/asimovs-chronology-of-science-and-discovery/",0,"",""],["Goodhart Ethology","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/z2BPxcFfhKho89D8L/goodhart-ethology",0,"","goodharts-law"],["Immobile AI makes a move: anti-wireheading, ontology change, and model splintering","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZqSESJYcA8m2Hh7Qm/immobile-ai-makes-a-move-anti-wireheading-ontology-change",0,"","reward-hacking"],["Investigating AI Takeover Scenarios","Sammy Martin","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zkF9PNSyDKusoyLkP/investigating-ai-takeover-scenarios",0,"",""],["Is Curiosity All You Need? On the Utility of Emergent Behaviours from Curious Exploration","Oliver Groth and 7 others","2021","blog","deepmind.com","www.deepmind.com/blog/is-curiosity-all-you-need-on-the-utility-of-emergent-behaviours-from-curious-exploration",0,"",""],["The theory-practice gap","Buck","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xRyLxfytmLFZ6qz5s/the-theory-practice-gap",0,"","scalable-oversight benchmarks"],["ThriftyDAgger: Budget-Aware Novelty and Risk Gating for Interactive Imitation Learning","Ryan Hoque and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.08273",0,"","rlhf policy"],["Counterfactual Contracts","harsimony","2021","blog","LessWrong","www.lesswrong.com/posts/BRHAWp7T3srrcuWDS/counterfactual-contracts",0,"","theory"],["Economic AI Safety","jsteinhardt","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RLcEQtoc5EqTxdan8/economic-ai-safety",0,"",""],["How truthful is GPT-3? A benchmark for language models","Owain_Evans","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PF58wEdztZFX2dSue/how-truthful-is-gpt-3-a-benchmark-for-language-models",0,"","benchmarks"],["I wanted to interview Eliezer Yudkowsky but he's busy so I simulated him instead","lsusr","2021","blog","LessWrong","www.lesswrong.com/posts/bDMoMvw2PYgijqZCC/i-wanted-to-interview-eliezer-yudkowsky-but-he-s-busy-so-i",0,"",""],["Jitters No Evidence of Stupidity in RL","1a3orn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fx8gCJu5zuLdZezTN/jitters-no-evidence-of-stupidity-in-rl",0,"",""],["The Metaethics and Normative Ethics of AGI Value Alignment: Many Questions, Some Implications","Eleos Arete Citrini","2021","blog","LessWrong","www.lesswrong.com/posts/mh4LasKBaqdYymhB2/the-metaethics-and-normative-ethics-of-agi-value-alignment",0,"",""],["[AN #164]: How well can language models write code?","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/p8xcZerxHWi4nLorx/an-164-how-well-can-language-models-write-code",0,"",""],["Call to Vigilance","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/call-to-vigilance/",0,"",""],["Challenges in Detoxifying Language Models","Johannes Welbl and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.07445",0,"","evals"],["Challenges in Detoxifying Language Models","Johannes Welbl and 9 others","2021","blog","deepmind.com","www.deepmind.com/blog/challenges-in-detoxifying-language-models",0,"",""],["Oracle predictions don't apply to non-existent worlds","Chris_Leong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/psyhmuDhazzFJKjXf/oracle-predictions-don-t-apply-to-non-existent-worlds",0,"","theory"],["The Metaethics and Normative Ethics of AGI Value Alignment: Many Questions, Some Implications","Eleos Arete Citrini","2021","blog","EA Forum","forum.effectivealtruism.org/posts/sSWMWkiAHRdDdPrWN/the-metaethics-and-normative-ethics-of-agi-value-alignment",0,"",""],["Von Neumann–Morgenstern utility theorem","Wikipedia","2021","report","en.wikipedia.org","en.wikipedia.org/w/index.php?title=Von_Neumann%E2%80%93Morgenstern_utility_theorem&oldid=1044421624",0,"",""],["How to make the best of the most important century?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/making-the-best-of-the-most-important-century/",0,"",""],["How to make the best of the most important century?","Holden Karnofsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/Lbtcjfxhrs8kfKK2M/how-to-make-the-best-of-the-most-important-century-1",0,"","governance"],["Augmenting Decision Making via Interactive What-If Analysis","Sneha Gathani and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.06160",0,"","evals"],["DeepMind is hiring Long-term Strategy & Governance researchers","vishal","2021","blog","EA Forum","forum.effectivealtruism.org/posts/atbonGDAFegfeDbTF/deepmind-is-hiring-long-term-strategy-and-governance",0,"","governance"],["A Socially Aware Reinforcement Learning Agent for The Single Track Road Problem","Ido Shapira and Amos Azaria","2021","paper","arXiv preprint","arxiv.org/abs/2109.05486",0,"","agents"],["AI timelines and theoretical understanding of deep learning","Venky1024","2021","blog","EA Forum","forum.effectivealtruism.org/posts/aaAf5fda88QgG4YkB/ai-timelines-and-theoretical-understanding-of-deep-learning",0,"","forecasting"],["do not form your own opinion","Tamsin Leake","2021","blog","carado.moe","carado.moe/do-not-form-your-own-opinion.html",0,"",""],["Chris Olah on working at top AI labs without an undergrad degree","80000_Hours","2021","blog","EA Forum","forum.effectivealtruism.org/posts/DxNSjxcMhFnHLztcN/chris-olah-on-working-at-top-ai-labs-without-an-undergrad",0,"","interpretability"],["Measurement, Optimization, and Take-off Speed","jsteinhardt","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/u4mFjCPviXHAPjZK7/measurement-optimization-and-take-off-speed",0,"",""],["Paths To High-Level Machine Intelligence","Daniel_Eth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/amK9EqxALJXyd9Rb2/paths-to-high-level-machine-intelligence",0,"","forecasting"],["The Blackwell order as a formalization of knowledge","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wEjozSY9rhkpAaABt/the-blackwell-order-as-a-formalization-of-knowledge",0,"",""],["What is the EU AI Act and why should you care about it?","MathiasKB","2021","blog","EA Forum","forum.effectivealtruism.org/posts/bd7yr3eozzzhMuKCi/what-is-the-eu-ai-act-and-why-should-you-care-about-it",0,"","evals governance policy"],["A mesa-optimization perspective on AI valence and moral patienthood","jacobpfau","2021","blog","EA Forum","forum.effectivealtruism.org/posts/6LDXiJ5Er6nrAfBiN/a-mesa-optimization-perspective-on-ai-valence-and-moral",0,"","robustness"],["Bootstrapped Meta-Learning","Sebastian Flennerhag and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.04504",0,"","benchmarks agents"],["Countably Factored Spaces","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QEfbg6vbjGgfFzJM4/countably-factored-spaces",0,"",""],["One Cold Link: “The Past and Future of Economic Growth: A Semi-Endogenous Perspective”","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/past-and-future-of-economic-growth-paper/",0,"",""],["The alignment problem in different capability regimes","Buck","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HHunb8FPnhWaDAQci/the-alignment-problem-in-different-capability-regimes",0,"",""],["User Tampering in Reinforcement Learning Recommender Systems","Charles Evans and Atoosa Kasirzadeh","2021","paper","arXiv preprint","arxiv.org/abs/2109.04083",0,"","instrumental-convergence agents policy"],["[AN #163]: Using finite factored sets for causal and temporal inference","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/9Hxa6pxRrxkwjBKib/an-163-using-finite-factored-sets-for-causal-and-temporal",0,"",""],["Distinguishing AI takeover scenarios","Sam Clarke and Sammy Martin","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qYzqDtoQaZ3eDDyxa/distinguishing-ai-takeover-scenarios",0,"","agents"],["Gradient descent is not just more efficient genetic algorithms","leogao","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/c9NSeCapaKtP6kvQD/gradient-descent-is-not-just-more-efficient-genetic",0,"",""],["Medium Lights","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/medium-lights/",0,"",""],["AI Timelines: Where the Arguments, and the \"Experts,\" Stand","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/where-ai-forecasting-stands-today/",0,"","forecasting"],["AI Timelines: Where the Arguments, and the \"Experts,\" Stand","Holden Karnofsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/7JxsXYDuqnKMqa6Eq/ai-timelines-where-the-arguments-and-the-experts-stand",0,"","forecasting"],["Alignment via manually implementing the utility function","Chantiel","2021","blog","LessWrong","www.lesswrong.com/posts/7PxkMMKuNRyAdufCK/alignment-via-manually-implementing-the-utility-function",0,"",""],["It takes 5 layers and 1000 artificial neurons to simulate a single biological neuron [Link]","MichaelStJules","2021","blog","EA Forum","forum.effectivealtruism.org/posts/h7Rj8Y8YWZccYMy5J/it-takes-5-layers-and-1000-artificial-neurons-to-simulate-a",0,"","forecasting"],["Multi-Agent Inverse Reinforcement Learning: Suboptimal Demonstrations and Alternative Solution Concepts","sage_bergerson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pRD5u2omuDoMTuH39/multi-agent-inverse-reinforcement-learning-suboptimal",0,"","agents"],["List of AI safety courses and resources","Daniel del Castillo and 2 others","2021","blog","EA Forum","forum.effectivealtruism.org/posts/XvWWfq9iqFj8x7Eu8/list-of-ai-safety-courses-and-resources",0,"",""],["Obstacles to gradient hacking","leogao","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/KfX7Ld7BeCMQn5gbz/obstacles-to-gradient-hacking",0,"",""],["Crazy ideas sometimes do work","Aryeh Englander","2021","blog","EA Forum","forum.effectivealtruism.org/posts/yPoamJhaBMCErPT5s/crazy-ideas-sometimes-do-work",0,"",""],["How to get more academics enthusiastic about doing AI Safety research?","PabloAMC","2021","blog","EA Forum","forum.effectivealtruism.org/posts/LgmCk9Lzpiot6G4Xa/how-to-get-more-academics-enthusiastic-about-doing-ai-safety",0,"",""],["Robust fine-tuning of zero-shot models","Mitchell Wortsman and 10 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.01903",0,"","robustness"],["Finetuned Language Models Are Zero-Shot Learners","Jason Wei and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.01652",0,"","evals"],["Thoughts on gradient hacking","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/egzqHKkzhuZuivHZ4/thoughts-on-gradient-hacking",0,"",""],["A Gentle Introduction to Graph Neural Networks","Adam Pearce","2021","report","Distill","distill.pub/2021/gnn-intro",0,"",""],["Cold Links: Useful","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-useful/",0,"",""],["Competent Preferences","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7kuhXtwFdXvD2Ngie/competent-preferences",0,"","goodharts-law"],["Formalizing Objections against Surrogate Goals","VojtaKovarik","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/K4FrKRTrmyxrw5Dip/formalizing-objections-against-surrogate-goals",0,"",""],["Is there a name for the theory that \"There will be fast takeoff in real-world capabilities because almost everything is AGI-complete\"?","David Scott Krueger (formerly: capybaralet)","2021","blog","LessWrong","www.lesswrong.com/posts/d8BW4pBwT9sBrJ44m/is-there-a-name-for-the-theory-that-there-will-be-fast",0,"","forecasting"],["Understanding Convolutions on Graphs","Ameya Daigavane and 2 others","2021","report","Distill","distill.pub/2021/understanding-gnns",0,"",""],["AI Education in China and the United States","Dahlia Peterson and 2 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ai-education-in-china-and-the-united-states/",0,"",""],["CHAI Newsletter #2 2021","CHAI","2021","report","drive.google.com","drive.google.com/file/d/1NXlJccLPk2UV_Z3qOzldBXId4l6dEcHp/view?usp=sharing",0,"",""],["NIST AI Risk Management Framework request for information (RFI)","Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/uFLCwj6jcvnvMtBk3/nist-ai-risk-management-framework-request-for-information",0,"",""],["Problem Learning: Towards the Free Will of Machines","Yongfeng Zhang","2021","paper","arXiv preprint","arxiv.org/abs/2109.00177",0,"","robustness"],["Robot Hacking Games","Dakota Cary","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/robot-hacking-games/",0,"",""],["Small Data’s Big AI Potential","Husanjot Chahal and 2 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/small-datas-big-ai-potential/",0,"",""],["The DOD’s Hidden Artificial Intelligence Workforce","Diana Gehlhaus and 4 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/the-dods-hidden-artificial-intelligence-workforce/",0,"",""],["August 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/08/31/august-2021-newsletter/",0,"",""],["Call for research on evaluating alignment (funding + advice available)","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7Rvctxk73BrKqEaqh/call-for-research-on-evaluating-alignment-funding-advice",0,"","evals"],["Finite Factored Sets: Applications","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yGFiw23pJ32obgLbw/finite-factored-sets-applications",0,"",""],["Finite Factored Sets: Inferring Time","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hePucCfKyiRHECz3e/finite-factored-sets-inferring-time",0,"",""],["Forecasting transformative AI: the \"biological anchors\" method in a nutshell","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/forecasting-transformative-ai-the-biological-anchors-method-in-a-nutshell/",0,"","forecasting"],["Grokking the Intentional Stance","jbkjr","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jHSi6BwDKTLt5dmsG/grokking-the-intentional-stance",0,"","agents robustness"],["Reward splintering as reverse of interpretability","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/JZEpqrLh2xHx2xfAd/reward-splintering-as-reverse-of-interpretability",0,"","interpretability"],["The Telephone Theorem: Information At A Distance Is Mediated By Deterministic Constraints","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jJf4FrfiQdDGg7uco/the-telephone-theorem-information-at-a-distance-is-mediated",0,"",""],["What are biases, anyway? Multiple type signatures","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Jute9YcbYvm4ZWdXk/what-are-biases-anyway-multiple-type-signatures",0,"",""],["\"Epistemic maps\" for AI Debates? (or for other issues)","Harrison Durland","2021","blog","EA Forum","forum.effectivealtruism.org/posts/s33LLoR6vbwiyRpTm/epistemic-maps-for-ai-debates-or-for-other-issues",0,"",""],["A short introduction to machine learning","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qE73pqxAZmeACsAdF/a-short-introduction-to-machine-learning",0,"",""],["Alignment Research = Conceptual Alignment Research + Applied Alignment Research","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2Xfv3GQgo2kGER8vA/alignment-research-conceptual-alignment-research-applied",0,"",""],["∀V: A Utopia For Ever","Tamsin Leake","2021","blog","carado.moe","carado.moe/∀V.html",0,"",""],["Making of #IAN","kirchner.jan","2021","blog","EA Forum","forum.effectivealtruism.org/posts/JtaHnmWDsYGiaNn3a/making-of-ian",0,"",""],["The Governance Problem and the \"Pretty Good\" X-Risk","Zach Stein-Perlman","2021","blog","LessWrong","www.lesswrong.com/posts/hWDQEeZXYH5dN6Pzq/the-governance-problem-and-the-pretty-good-x-risk",0,"","governance"],["Brain-Computer Interfaces and AI Alignment","niplav","2021","blog","LessWrong","www.lesswrong.com/posts/rpRsksjrBXEDJuHHy/brain-computer-interfaces-and-ai-alignment",0,"",""],["What are good alignment conference papers?","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cSNaxb8wu564x9n6r/what-are-good-alignment-conference-papers",0,"","robustness"],["Why and How Governments Should Monitor AI Development","Jess Whittlestone and Jack Clark","2021","paper","arXiv preprint","arxiv.org/abs/2108.12427",0,"","governance policy monitoring"],["[AN #162]: Foundation models: a paradigm shift within AI","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Haawpd5rZrzkzvYRC/an-162-foundation-models-a-paradigm-shift-within-ai",0,"",""],["Can you control the past?","Joe Carlsmith","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PcfHSSAMNFMgdqFyB/can-you-control-the-past",0,"","theory"],["More on “multiple world-size economies per atom”","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/more-on-multiple-world-size-economies-per-atom/",0,"",""],["Introduction to Reducing Goodhart","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RozggPiqQxzzDaNYF/introduction-to-reducing-goodhart",0,"","goodharts-law"],["The gloves are off, the pants are on","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/the-gloves-are-off-the-pants-are-on/",0,"",""],["(apologies for Alignment Forum server outage last night)","Ruby","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rxMmhJbNbuywFqcBk/apologies-for-alignment-forum-server-outage-last-night",0,"",""],["MIRI/OP exchange about decision theory","Rob Bensinger","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FBbHEjkZzdupcjkna/miri-op-exchange-about-decision-theory-1",0,"","theory"],["Reasoning about Counterfactuals and Explanations: Problems, Results and Directions","Leopoldo Bertossi","2021","paper","arXiv preprint","arxiv.org/abs/2108.11004",0,"",""],["What are the top priorities in a slow-takeoff, multipolar world?","JP Addison","2021","blog","EA Forum","forum.effectivealtruism.org/posts/FHav3yN9uFpFxYiYx/what-are-the-top-priorities-in-a-slow-takeoff-multipolar",0,"","forecasting"],["Are we \"trending toward\" transformative AI? (How would we know?)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/are-we-trending-toward-transformative-ai-how-would-we-know/",0,"",""],["Extraction of human preferences 👨→🤖","arunraja-hub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PZYD5kBpeHWgE5jX4/extraction-of-human-preferences",0,"",""],["Extraction of human preferences 👨→🤖","arunraja-hub","2021","blog","LessWrong","www.lesswrong.com/posts/PZYD5kBpeHWgE5jX4/extraction-of-human-preferences",0,"",""],["The Codex Skeptic FAQ","Michaël Trazzi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Rhg27MqkxJsnZwoYg/the-codex-skeptic-faq",0,"",""],["Welcome & FAQ!","Ruby and habryka","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Yp2vYb4zHXEeoTkJc/welcome-and-faq",0,"",""],["what happens when you die?","Tamsin Leake","2021","blog","carado.moe","carado.moe/what-happens-when-you-die.html",0,"",""],["Yet More Modal Combat","Donald Hobson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EPMnjRdhHNd65XpDt/yet-more-modal-combat",0,"",""],["AI Risk for Epistemic Minimalists","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8fpzBHt7e6n7Qjoo9/ai-risk-for-epistemic-minimalists",0,"",""],["right to death, therefore","Tamsin Leake","2021","blog","carado.moe","carado.moe/right-to-death-therefore.html",0,"",""],["AI Safety Papers: An App for the TAI Safety Database","ozziegooen","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GgusnG2tiPEa4aYFS/ai-safety-papers-an-app-for-the-tai-safety-database",0,"",""],["Designing a Combinatorial Financial Options Market","Xintong Wang and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.06443",0,"","evals deception agents"],["From Optimizing Engagement to Measuring Value","Smitha Milli and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2008.12623",0,"","evals"],["Learning Causal Models of Autonomous Agents using Interventions","Pulkit Verma and Siddharth Srivastava","2021","paper","arXiv preprint","arxiv.org/abs/2108.09586",0,"","interpretability evals agents"],["[AN #161]: Creating generalizable reward functions for multiple tasks by learning a model of functional similarity","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wMCbo7HX3cFbtHZcM/an-161-creating-generalizable-reward-functions-for-multiple",0,"",""],["Analogies and General Priors on Intelligence","riceissa and Sammy Martin","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yFQkFNCszoJPZTnK6/analogies-and-general-priors-on-intelligence",0,"","forecasting"],["Asleep at the Keyboard? Assessing the Security of GitHub Copilot's Code Contributions","Hammond Pearce and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.09293",0,"","evals"],["How DeepMind's Generally Capable Agents Were Trained","1a3orn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DreKBuMvK7fdESmSJ/how-deepmind-s-generally-capable-agents-were-trained",0,"","agents"],["Provide feedback on Open Philanthropy’s AI alignment RFP","abergal and Nick_Beckstead","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/TmhrC93mj2Pgsox9t/provide-feedback-on-open-philanthropy-s-ai-alignment-rfp",0,"",""],["Safe Transformative AI via a Windfall Clause","Paolo Bova and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.09404",0,"","policy robustness"],["Cold Links: heartwarming sports stuff","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/cold-links-heartwarming-sports-stuff/",0,"",""],["A Framework for Understanding AI-Induced Field Change: How AI Technologies are Legitimized and Institutionalized","Benjamin Cedric Larsen","2021","paper","In Proceedings of the 2021 AAAI ACM Conference on AI Ethics and\n  Society","arxiv.org/abs/2108.07804",0,"","deception agents governance"],["Updates and Lessons from AI Forecasting","Jacob Steinhardt","2021","report","bounded-regret.ghost.io","bounded-regret.ghost.io/ai-forecasting/",0,"","forecasting robustness"],["Finite Factored Sets: Polynomials and Probability","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jr5kyRhNriCX2Ayyg/finite-factored-sets-polynomials-and-probability",0,"",""],["Forecasting transformative AI: what's the burden of proof?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/forecasting-transformative-ai-whats-the-burden-of-proof/",0,"","forecasting"],["1h-volunteers needed for a small AI Safety-related research project","PabloAMC","2021","blog","EA Forum","forum.effectivealtruism.org/posts/hwTdJToSsb4DxW2a2/1h-volunteers-needed-for-a-small-ai-safety-related-research",0,"",""],["1h-volunteers needed for a small AI Safety-related research project","PabloAMC","2021","blog","LessWrong","www.lesswrong.com/posts/xzYRbFYrkiuuvD6GJ/1h-volunteers-needed-for-a-small-ai-safety-related-research",0,"",""],["Downstream Evaluations of Rotary Position Embeddings","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/rotary-embeddings-eval-harness/",0,"","evals"],["Is it worth making a database for moral predictions?","Jonas Hallgren","2021","blog","LessWrong","www.lesswrong.com/posts/m64joiCkrCy9MhPsk/is-it-worth-making-a-database-for-moral-predictions",0,"",""],["Modelling Transformative AI Risks (MTAIR) Project: Introduction","Davidmanheim and Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qnA6paRwMky3Q6ktk/modelling-transformative-ai-risks-mtair-project-introduction",0,"",""],["On the Opportunities and Risks of Foundation Models","Rishi Bommasani and 39 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.07258",0,"",""],["Program Synthesis with Large Language Models","Jacob Austin and 10 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.07732",0,"",""],["Decision Transformer: Reinforcement Learning via Sequence Modeling.","Lili Chen and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.01345",0,"",""],["Designing Recommender Systems to Depolarize.","Jonathan Stray","2021","paper","arXiv preprint","arxiv.org/abs/2107.04953",0,"",""],["Feature Expansive Reward Learning: Rethinking Human Input.","Andreea Bobu and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2006.13208",0,"","rlhf evals deception agents"],["kolmogorov complexity objectivity and languagespace","Tamsin Leake","2021","blog","carado.moe","carado.moe/kolmogorov-objectivity-in-languagespace.html",0,"",""],["Measuring Coding Challenge Competence With APPS.","Dan Hendrycks and 10 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.09938",0,"","evals benchmarks"],["Unsolved Problems in ML Safety.","Dan Hendrycks and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.13916",0,"","robustness monitoring"],["A neural circuit for flexible control of persistent behavioral states.","Ni Ji and 7 others","2021","report","elifesciences.org","elifesciences.org/articles/62889",0,"","mechanistic-interpretability"],["A Robust Control Framework for Human Motion Prediction.","Andrea Bajcsy and 6 others","2021","report","ieeexplore.ieee.org","ieeexplore.ieee.org/abstract/document/9210199",0,"",""],["Agent-aware state estimation for autonomous vehicles.","Shane Parr and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.00366",0,"","agents"],["Agnostic Learning with Unknown Utilities.","Kush Bhatia and 5 others","2021","report","drops.dagstuhl.de","drops.dagstuhl.de/opus/volltexte/2021/13594/",0,"",""],["AMP: Adversarial Motion Priors for Stylized Physics-Based Character Control.","Xue Bin Peng and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.02180",0,"",""],["Analyzing Human Models that Adapt Online.","Andrea Bajcsy and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.05746",0,"","deception"],["Approaches to gradient hacking","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/S2jsBsZvqjBZa3pKT/approaches-to-gradient-hacking",0,"",""],["APS: Active Pretraining with Successor Features.","Hao Liu and Pieter Abbeel","2021","paper","arXiv preprint","arxiv.org/abs/2108.13956",0,"","evals benchmarks agents policy"],["Behavior From the Void: Unsupervised Active Pre-Training.","Hao Liu and Pieter Abbeel","2021","paper","arXiv preprint","arxiv.org/abs/2103.04551",0,"","evals"],["Beyond Engagement: Aligning Algorithmic Recommendations With Prosocial Goals.","Jonathan Stray","2021","report","partnershiponai.org","partnershiponai.org/beyond-engagement-aligning-algorithmic-recommendations-with-prosocial-goals/",0,"",""],["book recommendation: Greg Egan's","Tamsin Leake","2021","blog","carado.moe","carado.moe/greg-egan-axiomatic.html",0,"",""],["Building efficient, reliable, and ethical autonomous systems.","Justin Svegliato","2021","report","semanticscholar.org","www.semanticscholar.org/paper/Building-Efficient%2C-Reliable%2C-and-Ethical-Systems-Svegliato/1da1f40e379c4dd4303f0b769264191339afafea",0,"","agents"],["Causal Inference Struggles with Agency on Online Platforms.","Smitha Milli and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.08995",0,"",""],["Clusterability in Neural Networks.","Daniel Filan and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.03386",0,"",""],["Contrastive Code Representation Learning.","Paras Jain and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2007.04973",0,"",""],["CUAD: An Expert-Annotated NLP Dataset for Legal Contract Review.","Dan Hendrycks and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.06268",0,"","benchmarks training-data"],["Decoupling Representation Learning from Reinforcement Learning.","Adam Stooke and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2009.08319",0,"","benchmarks agents policy"],["Dynamically Switching Human Prediction Models for Efficient Planning.","Arjun Sripathy and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.07815",0,"",""],["Efficient Dynamics Estimation With Adaptive Model Sets.","Ellis Ratner; Andrea Bajcsy; Terrence Fong; Claire J and 2 others","2021","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/9357896",0,"",""],["Estimating and Penalizing Preference Shift in Recommender Systems.","Micah Carroll and 3 others","2021","report","dl.acm.org","dl.acm.org/doi/abs/10.1145/3460231.3478849",0,"",""],["Ethically compliant planning within moral communities.","Samer B Nashed and 2 others","2021","report","justinsvegliato.com","justinsvegliato.com/pdf/NSZaies21.pdf",0,"",""],["Ethically compliant sequential decision making.","Justin Svegliato and 2 others","2021","report","aaai.org","www.aaai.org/AAAI21Papers/AAAI-3534.SvegliatoJ.pdf",0,"",""],["Evaluating Strategy Exploration in Empirical Game-Theoretic Analysis.","Yongzhao Wang and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.10423",0,"","evals"],["Evolution Strategies for Approximate Solution of Bayesian Games.","Zun Li and Michael P Wellman","2021","report","researchgate.net","www.researchgate.net/profile/Zun-Li-2/publication/352017840_Evolution_Strategies_for_Approximate_Solution_of_Bayesian_Games/links/60b5b6594585154e5ef5b2ef/Evolution-Strategies-for-Approximate-Solution-of-Bayesian-Games.pdf",0,"",""],["Explaining robot policies.","Olivia Watkins and 9 others","2021","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/10.1002/ail2.52",0,"",""],["Explore and Control with Adversarial Surprise.","Arnaud Fickinger and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.07394",0,"","deception agents"],["Improving Competence via Iterative State Space Refinement.","Connor Basich and 6 others","2021","report","ieeexplore.ieee.org","ieeexplore.ieee.org/abstract/document/9636239",0,"",""],["Improving Computational Efficiency in Visual Reinforcement Learning via Stored Embeddings.","Lili Chen and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.02886",0,"","agents policy"],["Iterative Empirical Game Solving via Single Policy Best Response.","Max Olan Smith and 2 others","2021","paper","ICLR 2021","arxiv.org/abs/2106.01901",0,"","agents policy"],["Learning State Representations from Random Deep Action-Conditional Predictions.","Zeyu Zheng and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.04897",0,"","robustness"],["Mapping the Political Economy of Reinforcement Learning Systems: The Case of Autonomous Vehicles.","Thomas Krendl Gilbert","2021","report","simons.berkeley.edu","simons.berkeley.edu/news/mapping-political-economy-reinforcement-learning-systems-case-autonomous-vehicles",0,"",""],["Mastering Atari Games with Limited Data.","Weirui Ye and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.00210",0,"","benchmarks"],["Measuring mathematical problem solving with the math dataset.","Dan Hendrycks and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.03874",0,"","robustness training-data"],["MSA Transformer.","Roshan Rao and 8 others","2021","report","biorxiv.org","www.biorxiv.org/content/10.1101/2021.02.12.430858v1",0,"",""],["Offline-to-Online Reinforcement Learning via Balanced Replay and Pessimistic Q-Ensemble.","Seunghyun Lee and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.00591",0,"","agents policy robustness"],["On complementing end-to-end human behavior predictors with planning.","Liting Sun and 3 others","2021","paper","Robotics: Science and Systems, 2021","arxiv.org/abs/2103.05661",0,"","robustness"],["On the benefits of randomly adjusting anytime weighted A*.","Abhinav Bhatia and 2 others","2021","report","justinsvegliato.com","justinsvegliato.com/pdf/BSZsocs21.pdf",0,"",""],["On the Expressivity of Markov Reward.","David Abel and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2111.00876",0,"","agents"],["Optimal Cost Design for Model Predictive Control.","Avik Jain and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.11353",0,"",""],["Passive Attention in Artificial Neural Networks Predicts Human Visual Selectivity.","Thomas A and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.07013",0,"","interpretability evals"],["Perceptual Adversarial Robustness: Defense Against Unseen Threat Models.","Cassidy Laidlaw and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2006.12655",0,"","robustness"],["Physical interaction as communication: Learning robot objectives online from human corrections.","Dylan P and 6 others","2021","report","journals.sagepub.com","journals.sagepub.com/doi/10.1177/02783649211050958",0,"",""],["Policy Gradient Bayesian Robust Optimization for Imitation Learning.","Zaynah Javed and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.06499",0,"","rlhf agents policy"],["Pragmatic Image Compression for Human-in-the-Loop Decision-Making.","Siddharth Reddy and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.04219",0,"","evals"],["Proper Value Equivalence.","Christopher Grimm and 4 others","2021","paper","NeurIPS 2021","arxiv.org/abs/2106.10316",0,"","deception policy robustness"],["Putting NeRF on a Diet: Semantically Consistent Few-Shot View Synthesis.","Ajay Jain and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.00677",0,"",""],["Reinforcement Learning of Implicit and Explicit Control Flow Instructions.","Ethan A and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.13195",0,"","agents"],["Reinforcement Learning with Latent Flow.","Wenling Shang and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.01857",0,"","evals benchmarks"],["Reward is Enough for Convex MDPs.","Tom Zahavy and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.00661",0,"","policy"],["Scalable Online Planning via Reinforcement Learning Fine-Tuning.","Arnaud Fickinger and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2109.15316",0,"","benchmarks policy"],["Show me the algorithm: Transparency in recommendation systems.","Jonathan Stray","2021","report","srinstitute.utoronto.ca","srinstitute.utoronto.ca/news/recommendation-systems-transparency",0,"","interpretability"],["Situational Confidence Assistance for Lifelong Shared Autonomy.","Matthew Zurek and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.06556",0,"","robustness"],["Skill Preferences: Learning to Extract and Execute Robotic Skills from Human Feedback.","Xiaofei Wang and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.05382",0,"","rlhf deception"],["State Entropy Maximization with Random Encoders for Efficient Exploration.","Younggyo Seo and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.09430",0,"","benchmarks"],["Unsupervised Learning of Visual 3D Keypoints for Control.","Boyuan Chen and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.07643",0,"","benchmarks deception"],["URLB: Unsupervised Reinforcement Learning Benchmark.","Michael Laskin and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.15191",0,"","evals benchmarks agents"],["Using metareasoning to maintain and restore safety for reliable autonomy.","Justin Svegliato and 2 others","2021","report","justinsvegliato.com","justinsvegliato.com/pdf/SBSZr2aw21.pdf",0,"",""],["[AN #160]: Building AIs that learn and think like people","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dkeDMktXtSjfoWnan/an-160-building-ais-that-learn-and-think-like-people",0,"",""],["A review of \"Agents and Devices\"","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/WrsQfBRqAPKiyGygT/a-review-of-agents-and-devices",0,"","agents"],["A few quick links re: COVID-19/Delta","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/a-few-quick-links-re-covid-19-delta/",0,"",""],["Power-seeking for successive choices","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5Hc4R6rj5yJ3xBhiX/power-seeking-for-successive-choices",0,"","instrumental-convergence power-seeking"],["Some criteria for sandwiching projects","dmz","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gfbf7RsE2fvxGXKC5/some-criteria-for-sandwiching-projects",0,"",""],["Automating Auditing: An ambitious concrete technical research proposal","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cQwT8asti3kyA62zc/automating-auditing-an-ambitious-concrete-technical-research",0,"","interpretability benchmarks deception robustness"],["Beyond Fairness Metrics: Roadblocks and Challenges for Ethical AI in Practice","Jiahao Chen and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.06217",0,"",""],["Give Sports a Chance","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/give-sports-a-chance/",0,"",""],["A Qualitative and Intuitive Explanation of Expected Value","Adam Zerner","2021","blog","LessWrong","www.lesswrong.com/posts/LJArjH2h4TACfksaT/a-qualitative-and-intuitive-explanation-of-expected-value",0,"","theory"],["A Rational Account of Anchor Effects in Hindsight Bias.","Samarie Wilson and 4 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/wilson_rational.pdf",0,"",""],["A rational model of people’s inferences about others’ preferences based on response times.","Vael Gates and 3 others","2021","report","psyarxiv.com","psyarxiv.com/25zfx",0,"",""],["A Strategic Analysis of Portfolio Compression.","Katherine Mayo and Michael P Wellman","2021","report","ifaamas.org","www.ifaamas.org/Proceedings/aamas2021/pdfs/p1599.pdf",0,"",""],["An Agent-Based Model of Strategic Adoption of Real-Time Payments.","Katherine Mayo and 3 others","2021","report","strategicreasoning.org","strategicreasoning.org/wp-content/uploads/2021/10/ICAIF_paper_108.pdf",0,"","agents"],["Asking the Right Questions: Learning Interpretable Action Models Through Query Answering.","Pulkit Verma and 2 others","2021","report","pulkitverma.net","pulkitverma.net/assets/pdf/vms_aaai21/vms_aaai21.pdf",0,"","interpretability"],["Evaluating models of robust word recognition with serial reproduction.","Stephan C and 4 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/meylanevaluating.pdf",0,"","evals"],["Extending rational models of communication from beliefs to actions.","Theodore R and 7 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/sumers_extending_2021.pdf",0,"",""],["Fixation patterns in simple choice reflect optimal information sampling.","Frederick Callaway and 3 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/callawayfixation2.pdf",0,"",""],["Forecasting Transformative AI, Part 1: What Kind of AI?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/transformative-ai-timelines-part-1-of-4-what-kind-of-ai/",0,"","forecasting"],["Forecasting Transformative AI: What Kind of AI?","Holden Karnofsky","2021","blog","EA Forum","forum.effectivealtruism.org/posts/AmxxnazJcBWzWEeqj/forecasting-transformative-ai-what-kind-of-ai",0,"","forecasting"],["From convolutional neural networks to models of higher level cognition (and back again).","Ruairidh M Battleday and 2 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/battledayfrom.pdf",0,"",""],["Hindsight Task Relabelling: Experience Replay for Sparse Reward Meta-RL.","Charles Packer and 3 others","2021","report","papers.nips.cc","papers.nips.cc/paper/2021/hash/1454ca2270599546dfcd2a3700e4d2f1-Abstract.html",0,"",""],["Human biases limit cumulative innovation.","Bill Thompson and Thomas L and Griffiths","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/thompsonhuman.pdf",0,"",""],["Human-Compatible Artificial Intelligence.","Stuart Russell","2021","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/mi19book-hcai.pdf",0,"",""],["Improving Transferability of Representations via Augmentation-Aware Self-Supervision.","Hankook Lee and 4 others","2021","report","papers.nips.cc","papers.nips.cc/paper/2021/file/94130ea17023c4837f0dcdda95034b65-Paper.pdf",0,"",""],["Intuitions about magic track the development of intuitive physics.","Casey Lewry and 5 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/lewry_intuitions.pdf",0,"",""],["Learning What To Do by Simulating the Past.","David Lindner and 3 others","2021","report","openreview.net","openreview.net/pdf?id=kBVJ2NtiY-",0,"",""],["Making Algorithms Work for Reporting.","Jonathan Stray","2021","report","library.oapen.org","library.oapen.org/bitstream/handle/20.500.12657/47509/9789048542079.pdf#page=146",0,"",""],["Meta-Learning of Structured Task Distributions in Humans and Machines.","Sreejan Kumar and 7 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/kumarmetalearning.pdf",0,"",""],["Quantifying Differences in Reward Functions.","Adam Gleave and 4 others","2021","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/iclr21-epic.pdf",0,"",""],["Replay-Guided Adversarial Environment Design.","Minqi Jiang and 5 others","2021","report","proceedings.neurips.cc","proceedings.neurips.cc/paper/2021/file/0e915db6326b6fb6a3c56546980a8c93-Paper.pdf",0,"",""],["Reward is Enough.","David Silver and 3 others","2021","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0004370221000862",0,"",""],["Serial reproduction reveals the geometry of visuospatial representations.","Thomas A and 6 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/langloisserial.pdf",0,"",""],["Spoofing the Limit Order Book: A Strategic Agent-Based Analysis.","Xintong Wang and 3 others","2021","report","mdpi.com","www.mdpi.com/2073-4336/12/2/46",0,"","agents"],["Stability Effects of Arbitrage in Exchange Traded Funds: An Agent-Based Model.","Megan Shearer and 3 others","2021","report","strategicreasoning.org","strategicreasoning.org/wp-content/uploads/2021/11/Megan_ICAIF_2021.pdf",0,"","agents"],["Teachable Reinforcement Learning via Advice Distillation.","Olivia Watkins and 4 others","2021","report","papers.nips.cc","papers.nips.cc/paper/2021/hash/37cfff3c04f95b22bcf166df586cd7a9-Abstract.html",0,"",""],["The Dynamics of Exemplar and Prototype Representations Depend on Environmental Statistics.","Arjun Devraj and 3 others","2021","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/devraj_dynamics.pdf",0,"",""],["The history and future of AI.","Stuart Russell","2021","report","academic.oup.com","academic.oup.com/oxrep/article-abstract/37/3/509/6374673",0,"",""],["Transforming Worlds: Automated Involutive MCMC for Open-Universe Probabilistic Models.","George Matheos and 7 others","2021","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/aabi21-oupm.pdf",0,"",""],["Tuning the hyperparameters of anytime planning: A deep reinforcement learning approach.","Abhinav Bhatia and 2 others","2021","report","openreview.net","openreview.net/pdf?id=c7hpFp_eRCo",0,"",""],["Unifying Principles and Metrics for Safe and Assistive AI.","Siddharth Srivastava","2021","report","aair-lab.github.io","aair-lab.github.io/Publications/srivastava_aaai21.pdf",0,"",""],["X2T: Training an X-to-Text Typing Interface with Online Learning from User Feedback.","Jensen Gao and 7 others","2021","report","openreview.net","openreview.net/forum?id=LiX3ECzDPHZ",0,"",""],["Goal-Directedness and Behavior, Redux","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YApiu7x3oTTzDgFFN/goal-directedness-and-behavior-redux",0,"",""],["When Most VNM-Coherent Preference Orderings Have Convergent Instrumental Incentives","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LYxWrxram2JFBaeaq/when-most-vnm-coherent-preference-orderings-have-convergent",0,"","instrumental-convergence"],["Applications for Deconfusing Goal-Directedness","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ECPmgwwWBikTtdqXo/applications-for-deconfusing-goal-directedness",0,"","instrumental-convergence"],["Seeking Power is Convergently Instrumental in a Broad Class of Environments","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hzeLSQ9nwDkPc4KNt/seeking-power-is-convergently-instrumental-in-a-broad-class",0,"","instrumental-convergence"],["DySR: A Dynamic Representation Learning and Aligning based Model for Service Bundle Recommendation","Mingyi Liu and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.03360",0,"","evals"],["Research agenda update","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DkfGaZTgwsE7XZq9k/research-agenda-update",0,"",""],["What 2026 looks like","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Xgy6CAf2jqHhynHL/what-2026-looks-like",0,"","forecasting"],["What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","Ajay Mandlekar and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.03298",0,"","evals benchmarks agents policy"],["Evaluating CLIP: Towards Characterization of Broader Capabilities and Downstream Implications","Sandhini Agarwal and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.02818",0,"","evals"],["Sharing the World with Digital Minds","Carl Shulman and Nick Bostrom","2021","report","oxford.universitypressscholarship.com","oxford.universitypressscholarship.com/view/10.1093/oso/9780192894076.001.0001/oso-9780192894076-chapter-18",0,"",""],["Traps of Formalization in Deconfusion","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pEB3LrNxvMKFLGBSG/traps-of-formalization-in-deconfusion",0,"",""],["Why talk about 10,000 years from now?","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/why-talk-about-10-000-years-from-now/",0,"",""],["[AN #159]: Building agents that know how to experiment, by training on procedurally generated games","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zvWqPmQasssaAWkrj/an-159-building-agents-that-know-how-to-experiment-by",0,"","agents"],["Chris Olah on what the hell is going on inside neural networks","80000_Hours","2021","blog","EA Forum","forum.effectivealtruism.org/posts/iZ6e2M4pmkNb3Dji5/chris-olah-on-what-the-hell-is-going-on-inside-neural",0,"","interpretability"],["Garrabrant and Shah on human modeling in AGI","Rob Bensinger","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Wap8sSDoiigrJibHA/garrabrant-and-shah-on-human-modeling-in-agi",0,"","scalable-oversight interpretability agents"],["July 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/08/03/july-2021-newsletter/",0,"",""],["The Great Depression, Recession and Stagnation in Full Historical Context","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/the-great-stagnation-in-full-historical-context/",0,"",""],["Value loading in the human brain: a worked example","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/iMM6dvHzco6jBMFMX/value-loading-in-the-human-brain-a-worked-example",0,"",""],["With One Voice: Composing a Travel Voice Assistant from Re-purposed Models","Shachaf Poran and 3 others","2021","paper","2nd International Workshop on Industrial Recommendation Systems @\n  KDD 2021","arxiv.org/abs/2108.11463",0,"",""],["How Do AI Timelines Affect Giving Now vs. Later?","MichaelDickens","2021","blog","EA Forum","forum.effectivealtruism.org/posts/bxvzu7qBF4cAsSu6d/how-do-ai-timelines-affect-giving-now-vs-later",0,"","forecasting"],["How should my timelines influence my career choice?","Tom Lieberum","2021","blog","LessWrong","www.lesswrong.com/posts/Cwoerjzjw7p2GFJPS/how-should-my-timelines-influence-my-career-choice",0,"","forecasting"],["LCDT, A Myopic Decision Theory","adamShimi and evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y76durQHrfqwgwM5o/lcdt-a-myopic-decision-theory",0,"","deception theory"],["This Can't Go On","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/this-cant-go-on/",0,"",""],["Triggering Failures: Out-Of-Distribution detection by learning from local adversarial attacks in Semantic Segmentation","Victor Besnier and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.01634",0,"","robustness training-data"],["What does GPT-3 understand? Symbol grounding and Chinese rooms","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ns95FHkkzpjXh4x5Q/what-does-gpt-3-understand-symbol-grounding-and-chinese",0,"",""],["Bridging the gap: the case for an ‘Incompletely Theorized Agreement’ on AI policy","Charlotte Stix and Matthijs M. Maas","2021","report","doi.org","doi.org/10.1007/s43681-020-00037-w",0,"","policy"],["Headline or Trend Line?","Margarita Konaev and 5 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/headline-or-trend-line/",0,"",""],["Indonesia’s AI Promise in Perspective","Kayla Goode and Heeu Millie Kim","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/indonesias-ai-promise-in-perspective/",0,"",""],["Military AI Cooperation Toolbox","Zoe Stanley-Lockman","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/military-ai-cooperation-toolbox/",0,"",""],["Reputations for Resolve and Higher-Order Beliefs in Crisis Bargaining","Allan Dafoe and 2 others","2021","report","doi.org","doi.org/10.1177/0022002721995549",0,"",""],["Responsible and Ethical Military AI","Zoe Stanley-Lockman","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/responsible-and-ethical-military-ai/",0,"",""],["Soft Calibration Objectives for Neural Networks","Archit Karandikar and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2108.00106",0,"","deception"],["Towards Understanding the Impact of Real-Time AI-Powered Educational Dashboards (RAED) on Providing Guidance to Instructors","Ajay Kulkarni","2021","paper","arXiv preprint","arxiv.org/abs/2107.14414",0,"",""],["[AN #158]: Should we be optimistic about generalization?","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/79qCdyfGxWNKbH8zk/an-158-should-we-be-optimistic-about-generalization",0,"",""],["An Ethical Framework for Guiding the Development of Affectively-Aware Artificial Intelligence","Desmond C. Ong","2021","paper","arXiv preprint","arxiv.org/abs/2107.13734",0,"","evals policy"],["Did they or didn't they learn tool use?","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8GoynCn4jaXKsiDky/did-they-or-didn-t-they-learn-tool-use",0,"","tool-use"],["How much compute was used to train DeepMind's generally capable agents?","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/KaPaTdpLggdMqzdyo/how-much-compute-was-used-to-train-deepmind-s-generally",0,"","agents"],["Imagining yourself as a digital person (two sketches)","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/imagining-yourself-as-a-digital-person-two-sketches/",0,"",""],["A Reflection on Learning from Data: Epistemology Issues and Limitations","Ahmad Hammoudeh and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.13270",0,"","deception"],["Discovering User-Interpretable Capabilities of Black-Box Planning Agents","Pulkit Verma and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.13668",0,"","interpretability evals agents"],["Does X cause Y? An in-depth evidence review","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/does-x-cause-y-an-in-depth-evidence-review/",0,"",""],["[3-hour podcast]: Joseph Carlsmith on longtermism, utopia, the computational power of the brain, meta-ethics, illusionism and meditation","Gus Docker","2021","blog","EA Forum","forum.effectivealtruism.org/posts/sdr3reRg7YT3kEnHX/3-hour-podcast-joseph-carlsmith-on-longtermism-utopia-the",0,"",""],["DeepMind: Generally capable agents emerge from open-ended play","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/mTGrrX8SZJ2tQDuqz/deepmind-generally-capable-agents-emerge-from-open-ended",0,"","agents"],["DeepMind: Generally capable agents emerge from open-ended play","kokotajlod","2021","blog","EA Forum","forum.effectivealtruism.org/posts/G43oe4JGfesBjFtTB/deepmind-generally-capable-agents-emerge-from-open-ended",0,"","agents"],["Digital People FAQ","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/digital-people-faq/",0,"",""],["Digital People Would Be An Even Bigger Deal","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/how-digital-people-could-change-the-world/",0,"",""],["Human-Level Reinforcement Learning through Theory-Based Modeling, Exploration, and Planning","Pedro A. Tsividis and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.12544",0,"","evals agents"],["Open-Ended Learning Leads to Generally Capable Agents","Open Ended Learning Team and 17 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.12808",0,"","evals agents tool-use robustness"],["Towards Industrial Private AI: A two-tier framework for data and model security","Sunder Ali Khowaja and 4 others","2021","paper","IEEE Wireless Communications 2022","arxiv.org/abs/2107.12806",0,"",""],["AMA: The new Open Philanthropy Technology Policy Fellowship","lukeprog","2021","blog","EA Forum","forum.effectivealtruism.org/posts/2sn8RWPaChvyuHCcp/ama-the-new-open-philanthropy-technology-policy-fellowship",0,"","governance policy"],["Refactoring Alignment (attempt #2)","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vayxfTSQEDtwhPGpW/refactoring-alignment-attempt-2",0,"",""],["what is value?","Tamsin Leake","2021","blog","carado.moe","carado.moe/what-is-value.html",0,"",""],["[AN #157]: Measuring misalignment in the technology underlying Copilot","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cyTP4ZMnN6RFu9L62/an-157-measuring-misalignment-in-the-technology-underlying",0,"","deception"],["AXRP Episode 10 - AI’s Future and Impacts with Katja Grace","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xbABZRxoSTAnsf8os/axrp-episode-10-ai-s-future-and-impacts-with-katja-grace",0,"","forecasting"],["Gallup website notes","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/gallup-website-notes/",0,"",""],["Standardized Max Logits: A Simple yet Effective Approach for Identifying Unexpected Road Obstacles in Urban-Scene Segmentation","Sanghun Jung and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.11264",0,"","benchmarks"],["Enabling high-accuracy protein structure prediction at the proteome scale","Kathryn Tunyasuvunakool and 32 others","2021","blog","deepmind.com","www.deepmind.com/blog/enabling-high-accuracy-protein-structure-prediction-at-the-proteome-scale",0,"",""],["Re-Define Intent Alignment?","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7fkaJLzRiEr2hmSDi/re-define-intent-alignment",0,"",""],["What are you optimizing for? Aligning Recommender Systems with Human Values","Jonathan Stray and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.10939",0,"",""],["Reward splintering for AI design","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xoQhHxgwdHvWhj4P4/reward-splintering-for-ai-design",0,"",""],["Track records for those who have made lots of predictions","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/prediction-track-records-i-know-of/",0,"",""],["Apply to the new Open Philanthropy Technology Policy Fellowship!","lukeprog","2021","blog","EA Forum","forum.effectivealtruism.org/posts/4H7j4PQjTDK4W6u79/apply-to-the-new-open-philanthropy-technology-policy",0,"","governance policy"],["culture tribes and legitimacy","Tamsin Leake","2021","blog","carado.moe","carado.moe/culture-tribes-legitimacy.html",0,"",""],["Entropic boundary conditions towards safe artificial superintelligence","Santiago Nunez-Corrales","2021","blog","LessWrong","www.lesswrong.com/posts/QDv3y88KkrroCazeB/entropic-boundary-conditions-towards-safe-artificial",0,"",""],["systems and diversity","Tamsin Leake","2021","blog","carado.moe","carado.moe/systems-and-diversity.html",0,"",""],["The Duplicator: Instant Cloning Would Make the World Economy Explode","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/the-duplicator/",0,"",""],["A Modulation Layer to Increase Neural Network Robustness Against Data Quality Issues","Mohamed Abdelhack and 7 others","2021","paper","Transactions on Machine Learning Research 2023","arxiv.org/abs/2107.08574",0,"","robustness"],["Is the argument that AI is an xrisk valid?","MACannon","2021","blog","LessWrong","www.lesswrong.com/posts/bsJH4uDSLxS3eAZeJ/is-the-argument-that-ai-is-an-xrisk-valid",0,"",""],["On the Veracity of Local, Model-agnostic Explanations in Audio Classification: Targeted Investigations with Adversarial Examples","Verena Praher and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.09045",0,"","evals robustness"],["A model of decision-making in the brain (the short version)","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/e5duEqhAhurT8tCyr/a-model-of-decision-making-in-the-brain-the-short-version",0,"",""],["Books and lecture series relevant to AI governance?","MichaelA","2021","blog","EA Forum","forum.effectivealtruism.org/posts/5LnyxoBZK7FQHPvi3/books-and-lecture-series-relevant-to-ai-governance",0,"","governance"],["botched alignment and alignment awareness","Tamsin Leake","2021","blog","carado.moe","carado.moe/botched-alignment-and-awareness.html",0,"",""],["AI alignment timeline codes","Tamsin Leake","2021","blog","carado.moe","carado.moe/timeline-codes.html",0,"","forecasting"],["when in doubt, kill everyone","Tamsin Leake","2021","blog","carado.moe","carado.moe/when-in-doubt-kill-everyone.html",0,"",""],["[AN #156]: The scaling hypothesis: a plan for building AGI","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/XusDPpXr6FYJqWkxh/an-156-the-scaling-hypothesis-a-plan-for-building-agi",0,"",""],["A personal take on longtermist AI governance","lukeprog","2021","blog","EA Forum","forum.effectivealtruism.org/posts/M2SBwctwC6vBqAmZW/a-personal-take-on-longtermist-ai-governance",0,"","governance forecasting"],["AI alignment and wolfram physics","Tamsin Leake","2021","blog","carado.moe","carado.moe/ai-alignment-wolfram-physics.html",0,"",""],["Bayesianism versus conservatism versus Goodhart","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EFZ64igiNNwiLHaYk/bayesianism-versus-conservatism-versus-goodhart",0,"","goodharts-law"],["Underlying model of an imperfect morphism","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RnxkAiGcQpfErjHYT/underlying-model-of-an-imperfect-morphism",0,"","deception"],["A closer look at chess scalings (into the past)","hippke","2021","blog","LessWrong","www.lesswrong.com/posts/4MLBK7iCW3vYd93Mn/a-closer-look-at-chess-scalings-into-the-past",0,"","scaling-laws"],["Collective Action on Artificial Intelligence: A Primer and Review | Global Catastrophic Risk Institute","Robert de Neufville","2021","report","gcrinstitute.org","gcrinstitute.org/collective-action-on-artificial-intelligence-a-primer-and-review/",0,"",""],["Fractional progress estimates for AI timelines and implied resource requirements","Mark Xu and CarlShulman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/h3ejmEeNniDNFXTgp/fractional-progress-estimates-for-ai-timelines-and-implied",0,"","forecasting"],["Generalizing Koopman-Pitman-Darmois","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tGCyRQigGoqA4oSRo/generalizing-koopman-pitman-darmois",0,"",""],["Highly accurate protein structure prediction with AlphaFold | Nature","John Jumper and 33 others","2021","report","nature.com","www.nature.com/articles/s41586-021-03819-2",0,"",""],["Phil Birnbaum's \"bad regression\" puzzles","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/phil-birnbaums-regression-analysis/",0,"",""],["universal complete","Tamsin Leake","2021","blog","carado.moe","carado.moe/universal-complete.html",0,"",""],["Conservative Objective Models for Effective Offline Model-Based Optimization","Brandon Trabucco and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.06882",0,"","robustness"],["Deep Adaptive Multi-Intention Inverse Reinforcement Learning","Ariyan Bighashdel and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.06692",0,"","evals benchmarks"],["Honesty about reading","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/honesty-about-reading/",0,"",""],["Melting Pot: an evaluation suite for multi-agent reinforcement learning","Joel Z. Leibo and 9 others","2021","blog","deepmind.com","www.deepmind.com/blog/melting-pot-an-evaluation-suite-for-multi-agent-reinforcement-learning",0,"","evals agents"],["Model-based RL, Desires, Brains, Wireheading","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/K5ikTdaNymfWXQHFb/model-based-rl-desires-brains-wireheading",0,"","reward-hacking"],["Scalable Evaluation of Multi-Agent Reinforcement Learning with Melting Pot","Joel Z. Leibo and 9 others","2021","paper","In International Conference on Machine Learning 2021 (pp.\n  6187-6199). PMLR","arxiv.org/abs/2107.06857",0,"","evals benchmarks agents policy training-data"],["The Benchmark Lottery","Mostafa Dehghani and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.07002",0,"","evals benchmarks robustness"],["All Possible Views About Humanity's Future Are Wild","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/all-possible-views-about-humanitys-future-are-wild/",0,"",""],["Answering questions honestly instead of predicting human answers: lots of problems and some solutions","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gEw8ig38mCGjia7dj/answering-questions-honestly-instead-of-predicting-human",0,"",""],["Doomsday and objective chance","Teruji Thomas","2021","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/doomsday-and-objective-chance-teruji-thomas/",0,"",""],["First Post","Holden Karnofsky","2021","blog","cold-takes.com","www.cold-takes.com/first-post/",0,"",""],["The Additive Summary Equation","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/E4GvMdELt6s6CaXrb/the-additive-summary-equation",0,"",""],["What will the twenties look like if AGI is 30 years away?","Daniel Kokotajlo","2021","blog","LessWrong","www.lesswrong.com/posts/axbTNGuMtB4hCkNus/what-will-the-twenties-look-like-if-agi-is-30-years-away",0,"","forecasting"],["A paradox for tiny probabilities and enormous values - Nick Beckstead (Open Philanthropy Project) and Teruji Thomas (Global Priorities Institute, Oxford University)","Nick Beckstead and Teruji Thomas","2021","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/nick-beckstead-and-teruji-thomas-a-paradox-for-tiny-probabilities-and-enormous-values/",0,"",""],["Anthropic decision theory for self-locating beliefs","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MZJxtzjSeezEkedWn/anthropic-decision-theory-for-self-locating-beliefs",0,"","theory"],["The inescapability of knowledge","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DLjCSHjwbxzEEa6Hu/the-inescapability-of-knowledge",0,"",""],["The More Power At Stake, The Stronger Instrumental Convergence Gets For Optimal Policies","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Yc5QSSZCQ9qdyxZF6/the-more-power-at-stake-the-stronger-instrumental",0,"","instrumental-convergence"],["The Role of Social Movements, Coalitions, and Workers in Resisting Harmful Artificial Intelligence and Contributing to the Development of Responsible AI","Susan von Struensee","2021","paper","arXiv preprint","arxiv.org/abs/2107.14052",0,"",""],["The accumulation of knowledge: literature review","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dkruhqAEhXnbAk7iJ/the-accumulation-of-knowledge-literature-review",0,"",""],["A Simple Model of AGI Deployment Risk","djbinder","2021","blog","EA Forum","forum.effectivealtruism.org/posts/aSMexrjGXpNiWpbb5/a-simple-model-of-agi-deployment-risk",0,"",""],["Aligning an optical interferometer with beam divergence control and continuous action space","Stepan Makarenko and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.04457",0,"","evals agents"],["estimating the amount of populated intelligence explosion timelines","Tamsin Leake","2021","blog","carado.moe","carado.moe/estimating-populated-intelligence-explosions.html",0,"","forecasting"],["Finite Factored Sets: Conditional Orthogonality","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hA6z9s72KZDYpuFhq/finite-factored-sets-conditional-orthogonality",0,"",""],["Generalised models: imperfect morphisms and informational entropy","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/mMCvmLMHXid5tHKju/generalised-models-imperfect-morphisms-and-informational",0,"",""],["Integrating Planning, Execution and Monitoring in the presence of Open World Novelties: Case Study of an Open World Monopoly Solver","Sriram Gopalakrishnan and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.04303",0,"","evals agents policy monitoring"],["The Centre for the Governance of AI is becoming a nonprofit","MarkusAnderljung","2021","blog","EA Forum","forum.effectivealtruism.org/posts/zcAxoAHcSECyewr2t/the-centre-for-the-governance-of-ai-is-becoming-a-nonprofit",0,"","governance"],["[AN #155]: A Minecraft benchmark for algorithms that learn without reward functions","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/a7YgzDYx4FhdB3TmR/an-155-a-minecraft-benchmark-for-algorithms-that-learn",0,"","benchmarks"],["A world in which the alignment problem seems lower-stakes","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/sunXMY5WyDcrHsNRr/a-world-in-which-the-alignment-problem-seems-lower-stakes",0,"","instrumental-convergence"],["Anthropics and Fermi: grabby, visible, zoo-keeping, and early aliens","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wgHbNZHsqfiXiqofd/anthropics-and-fermi-grabby-visible-zoo-keeping-and-early",0,"",""],["Anthropics in infinite universes","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rbJLrcmHtusGBudTY/anthropics-in-infinite-universes",0,"",""],["BASALT: A Benchmark for Learning from Human Feedback","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RyH8LtgMbRAJ9Dv6R/basalt-a-benchmark-for-learning-from-human-feedback",0,"","rlhf benchmarks"],["Intermittent Distillations #4: Semiconductors, Economics, Intelligence, and Technological Progress.","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rQGW2GqHAFprupYkf/intermittent-distillations-4-semiconductors-economics",0,"",""],["Intermittent Distillations #4: Semiconductors, Economics, Intelligence, and Technological Progress.","Mark Xu","2021","blog","LessWrong","www.lesswrong.com/posts/rQGW2GqHAFprupYkf/intermittent-distillations-4-semiconductors-economics",0,"",""],["Practical anthropics summary","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jfMExCKWipKeCdSuG/practical-anthropics-summary",0,"",""],["purposes for art","Tamsin Leake","2021","blog","carado.moe","carado.moe/purposes-for-art.html",0,"",""],["The SIA population update can be surprisingly small","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xfEsxAtBTLgFe7fSZ/the-sia-population-update-can-be-surprisingly-small",0,"",""],["A Decision Model for Decentralized Autonomous Organization Platform Selection: Three Industry Case Studies","Elena Baninemeh and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.14093",0,"","evals governance"],["A second example of conditional orthogonality in finite factored sets","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GFGNwCwkffBevyXR2/a-second-example-of-conditional-orthogonality-in-finite",0,"",""],["Agency and the unreliable autonomous car","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8AjDwHp9pvZdm6ZEp/agency-and-the-unreliable-autonomous-car",0,"",""],["Evaluating Large Language Models Trained on Code","Mark Chen and 39 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.03374",0,"","evals deception robustness training-data scaling-laws"],["How much chess engine progress is about adapting to bigger computers?","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/H6L7fuEN9qXDanQ6W/how-much-chess-engine-progress-is-about-adapting-to-bigger",0,"","scaling-laws"],["Not Quite 'Ask a Librarian': AI on the Nature, Value, and Future of LIS","Jesse David Dinneen and Helen Bubinger","2021","paper","arXiv preprint","arxiv.org/abs/2107.05383",0,"","evals forecasting"],["Quantifying curation","janus","2021","blog","generative.ink","generative.ink/posts/quantifying-curation/",0,"",""],["What A Long, Strange Trip It's Been: EleutherAI One Year Retrospective","Connor Leahy and 3 others","2021","blog","blog.eleuther.ai","blog.eleuther.ai/year-one/",0,"",""],["A simple example of conditional orthogonality in finite factored sets","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/qGjCt4Xq83MBaygPx/a-simple-example-of-conditional-orthogonality-in-finite",0,"",""],["An Orchestration Platform that Puts Radiologists in the Driver's Seat of AI Innovation: A Methodological Approach","Raphael Y. Cohen and Aaron D. Sodickson","2021","paper","arXiv preprint","arxiv.org/abs/2107.04409",0,"",""],["Getting started independently in AI Safety","JJ Hepburn","2021","blog","EA Forum","forum.effectivealtruism.org/posts/naJ9cJfHMTJ9CACvD/getting-started-independently-in-ai-safety",0,"",""],["Is keeping AI \"in the box\" during training enough?","tgb","2021","blog","LessWrong","www.lesswrong.com/posts/EFFQLG6qcBNfHS5M9/is-keeping-ai-in-the-box-during-training-enough",0,"",""],["ML-Quadrat & DriotData: A Model-Driven Engineering Tool and a Low-Code Platform for Smart IoT Services","Armin Moin and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.02692",0,"",""],["Anthropic Effects in Estimating Evolution Difficulty","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pg6Z5tiuXotGTWaG8/anthropic-effects-in-estimating-evolution-difficulty",0,"",""],["Corporate Governance of Artificial Intelligence in the Public Interest","Peter Cihon and 2 others","2021","report","mdpi.com","www.mdpi.com/2078-2489/12/7/275",0,"","governance"],["Logic Locking at the Frontiers of Machine Learning: A Survey on Developments and Opportunities","Dominik Sisejkovic and 4 others","2021","paper","2021 IFIP/IEEE 29th International Conference on Very Large Scale\n  Integration (VLSI-SoC)","arxiv.org/abs/2107.01915",0,"","mechanistic-interpretability evals"],["The MineRL BASALT Competition on Learning from Human Feedback","Rohin Shah and 12 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.01969",0,"","rlhf evals agents governance"],["Towards solving the 7-in-a-row game","Domonkos Czifra and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2107.05363",0,"",""],["Evolution as Backstop for Reinforcement Learning","Gwern Branwen","2021","blog","gwern.net","www.gwern.net/Backstop.page",0,"",""],["Mauhn Releases AI Safety Documentation","Berg Severens","2021","blog","LessWrong","www.lesswrong.com/posts/z4dna4cbvasn6BepA/mauhn-releases-ai-safety-documentation",0,"","governance"],["Confusions re: Higher-Level Game Theory","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FPML8k4QtjJxk3Y4M/confusions-re-higher-level-game-theory",0,"",""],["Distill Hiatus","Editorial Team","2021","report","Distill","distill.pub/2021/distill-hiatus",0,"",""],["June 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/07/01/june-2021-newsletter/",0,"",""],["AI Accidents: An Emerging Threat","Zachary Arnold and Helen Toner","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ai-accidents-an-emerging-threat/",0,"",""],["Experimentally evaluating whether honesty generalizes","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BxersHYN2qcFoonwg/experimentally-evaluating-whether-honesty-generalizes",0,"","evals agents policy robustness"],["National Power After AI","Matthew Daniels and Ben Chang","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/national-power-after-ai/",0,"",""],["[AN #154]: What economic growth theory has to say about transformative AI","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/9AcEy2zThvceT9kve/an-154-what-economic-growth-theory-has-to-say-about",0,"",""],["How to get technological knowledge on AI/ML (for non-tech people)","FangFang","2021","blog","EA Forum","forum.effectivealtruism.org/posts/FBtcr46GBiknNvWxy/how-to-get-technological-knowledge-on-ai-ml-for-non-tech",0,"","governance"],["Musings on general systems alignment","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hKMgCaAYS4hnanxBL/musings-on-general-systems-alignment",0,"",""],["Progress on Causal Influence Diagrams","tom4everitt","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cd7Hw492RqooYgQAS/progress-on-causal-influence-diagrams",0,"",""],["Progress on Causal Influence Diagrams","DeepMind Safety Research","2021","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/progress-on-causal-influence-diagrams-a7a32180b0d1",0,"","reward-hacking evals agents"],["The Threat of Offensive AI to Organizations","Yisroel Mirsky and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.15764",0,"",""],["Thoughts on safety in predictive learning","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ey7jACdF4j6GrQLrG/thoughts-on-safety-in-predictive-learning",0,"",""],["Iason Gabriel on Foundational Philosophical Questions in AI Alignment-by Future of Life Institute-video_id MzFl0SdjSso-date 20210630","Iason Gabriel","2021","report","drive.google.com","drive.google.com/file/d/1qM_XvyjdaXUQl2OXUW6CuEX3i5bBuP75/view?usp=share_link",0,"",""],["we're all doomed","Tamsin Leake","2021","blog","carado.moe","carado.moe/were-all-doomed.html",0,"",""],["disclosing subjectivity","Tamsin Leake","2021","blog","carado.moe","carado.moe/disclosing-subjectivity.html",0,"",""],["How teams went about their research at AI Safety Camp edition 5","Remmelt","2021","blog","LessWrong","www.lesswrong.com/posts/QEmfyhqMcSpfnY2dX/how-teams-went-about-their-research-at-ai-safety-camp",0,"",""],["Brute force searching for alignment","Donald Hobson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MRFXpedeKJRa324dL/brute-force-searching-for-alignment",0,"",""],["Finite Factored Sets: LW transcript with running commentary","Rob Bensinger and Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6t9F5cS3JjtSspbAZ/finite-factored-sets-lw-transcript-with-running-commentary",0,"",""],["[AN #153]: Experiments that demonstrate failures of objective robustness","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/u9CqcufkAJBwXdbx7/an-153-experiments-that-demonstrate-failures-of-objective",0,"","robustness"],["Anthropics and Embedded Agency","dadadarren","2021","blog","LessWrong","www.lesswrong.com/posts/jDTqKRdy3fxvc7fFH/anthropics-and-embedded-agency",0,"","theory"],["aiSTROM -- A roadmap for developing a successful AI strategy","Dorien Herremans","2021","paper","IEEE Access, 2021","arxiv.org/abs/2107.06071",0,"",""],["The positive case for a focus on achieving safe AI?","vipulnaik","2021","blog","EA Forum","forum.effectivealtruism.org/posts/yNxxtd8HAcEukCb8Z/the-positive-case-for-a-focus-on-achieving-safe-ai",0,"",""],["AXRP Episode 9 - Finite Factored Sets with Scott Garrabrant","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/s4FNjvrJG6zmYdBuG/axrp-episode-9-finite-factored-sets-with-scott-garrabrant",0,"","theory"],["classifying computational frameworks","Tamsin Leake","2021","blog","carado.moe","carado.moe/classifying-computational-frameworks.html",0,"",""],["degrees of runtime metaprogrammability","Tamsin Leake","2021","blog","carado.moe","carado.moe/degrees-of-runtime-metaprogrammability.html",0,"",""],["Modeling the Mistakes of Boundedly Rational Agents Within a Bayesian Theory of Mind","Arwa Alanqary and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.13249",0,"","agents"],["Shallow evaluations of longtermist organizations","NunoSempere","2021","blog","EA Forum","forum.effectivealtruism.org/posts/xmmqDdGqNZq5RELer/shallow-evaluations-of-longtermist-organizations",0,"","evals governance policy forecasting"],["AISC5: Research Summaries","Remmelt Ellen","2021","blog","aisafety.camp","aisafety.camp/2021/06/23/aisc5-research-summaries/",0,"",""],["Alex Turner's Research, Comprehensive Information Gathering","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rxsg2sTyHGnMTYbeH/alex-turner-s-research-comprehensive-information-gathering",0,"","instrumental-convergence"],["Discussion: Objective Robustness and Inner Alignment Terminology","jbkjr and Lauro Langosco","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pDaxobbB9FG5Dvqyv/discussion-objective-robustness-and-inner-alignment",0,"","robustness"],["Empirical Observations of Objective Robustness Failures","jbkjr and Lauro Langosco","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/iJDmL7HJtN5CYKReM/empirical-observations-of-objective-robustness-failures",0,"","interpretability benchmarks agents policy robustness"],["Frequent arguments about alignment","John Schulman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6ccG9i5cTncebmhsH/frequent-arguments-about-alignment",0,"","rlhf robustness"],["How Well do Feature Visualizations Support Causal Understanding of CNN Activations?","Roland S. Zimmermann and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.12447",0,"","interpretability mechanistic-interpretability"],["IQ-Learn: Inverse soft-Q Learning for Imitation","Divyansh Garg and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.12142",0,"","policy"],["New blog location","jsteinhardt","2021","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2021/06/23/new-blog-location/",0,"",""],["Not all users are the same: Providing personalized explanations for sequential decision making problems","Utkarsh Soni and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.12207",0,"","agents"],["Environmental Structure Can Cause Instrumental Convergence","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/b6jJddSvWMdZHJHh3/environmental-structure-can-cause-instrumental-convergence",0,"","instrumental-convergence power-seeking agents policy robustness"],["I’m no longer sure that I buy dutch book arguments and this makes me skeptical of the \"utility function\" abstraction","Eli Tyre","2021","blog","LessWrong","www.lesswrong.com/posts/ndFHYBZCCusq3Whb9/i-m-no-longer-sure-that-i-buy-dutch-book-arguments-and-this",0,"","theory"],["The scope of longtermism","David Thorstad","2021","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/the-scope-of-longtermism-david-thorstad-global-priorities-institute-university-of-oxford/",0,"",""],["cm21, a pixel art editor","Tamsin Leake","2021","blog","carado.moe","carado.moe/cm21.html",0,"",""],["Parameter counts in Machine Learning","Jsevillamol and Pablo Villalobos","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GzoWcYibWYwJva8aL/parameter-counts-in-machine-learning",0,"","scaling-laws"],["Uncertain Decisions Facilitate Better Preference Learning","Cassidy Laidlaw and Stuart Russell","2021","paper","arXiv preprint","arxiv.org/abs/2106.10394",0,"","deception policy robustness theory"],["Conditional offers and low priors: the problem with 1-boxing Newcomb's dilemma","Andrew Vlahos","2021","blog","LessWrong","www.lesswrong.com/posts/j6aa9aJgtdjr24WYj/conditional-offers-and-low-priors-the-problem-with-1-boxing",0,"",""],["Knowledge is not just precipitation of action","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/JMpERTz9TcnMfEapF/knowledge-is-not-just-precipitation-of-action",0,"",""],["MADE: Exploration via Maximizing Deviation from Explored Regions","Tianjun Zhang and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.10268",0,"","evals benchmarks policy"],["Non-poisonous cake: anthropic updates are normal","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3kwwDieE9SmFoXz9F/non-poisonous-cake-anthropic-updates-are-normal",0,"",""],["categories of knowledge representation","Tamsin Leake","2021","blog","carado.moe","carado.moe/categories-of-knowledge.html",0,"",""],["Poisoning and Backdooring Contrastive Learning","Nicholas Carlini","2021","paper","arXiv preprint","arxiv.org/abs/2106.09667",0,"","robustness training-data"],["Pros and cons of working on near-term technical AI safety and assurance","Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gBLs3GefMdtWe6iSk/pros-and-cons-of-working-on-near-term-technical-ai-safety",0,"","assurance"],["Thoughts on a \"Sequences Inspired\" PhD Topic","goose000","2021","blog","LessWrong","www.lesswrong.com/posts/J4wpcCTo6CF6C5ftB/thoughts-on-a-sequences-inspired-phd-topic",0,"",""],["[AN #152]: How we’ve overestimated few-shot learning capabilities","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/iCzGrppxQAJhRXhmD/an-152-how-we-ve-overestimated-few-shot-learning",0,"",""],["Aligning AI Regulation to Sociotechnical Change","Matthijs M. Maas","2021","report","papers.ssrn.com","papers.ssrn.com/abstract=3871635",0,"","governance"],["Developing a Fidelity Evaluation Approach for Interpretable Machine Learning","Mythreyi Velmurugan and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.08492",0,"","interpretability evals deception"],["Escaping the Löbian Obstacle","Morgan_Rogers","2021","blog","LessWrong","www.lesswrong.com/posts/gbNLvkGuGcmSFFpSE/escaping-the-loebian-obstacle",0,"","theory"],["Insufficient Values","Jozdien and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Pd53Mip7Aa3TsdA7E/insufficient-values",0,"",""],["Open problem: how can we quantify player alignment in 2x2 normal-form games?","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ghyw76DfRyiiMxo3t/open-problem-how-can-we-quantify-player-alignment-in-2x2",0,"",""],["Reward Is Not Enough","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/frApEhpyKQAcFvbXJ/reward-is-not-enough",0,"","agents"],["Futureproof: Artificial Intelligence Chapter | GovAI","Toby Ord and 5 others","2021","report","governance.ai","www.governance.ai/research-paper/futureproof-artificial-intelligence-chapter",0,"",""],["Knowledge is not just digital abstraction layers","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fcnFddKjKZdDXt5cp/knowledge-is-not-just-digital-abstraction-layers",0,"",""],["my answer to the fermi paradox","Tamsin Leake","2021","blog","carado.moe","carado.moe/fermi-paradox.html",0,"",""],["refusing to answer ≠ giving a negative answer","Tamsin Leake","2021","blog","carado.moe","carado.moe/refusing-negative.html",0,"",""],["Revisiting the Calibration of Modern Neural Networks","Matthias Minderer and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.07998",0,"","robustness"],["the many faces of chaos magick","Tamsin Leake","2021","blog","carado.moe","carado.moe/faces-chaos-magick.html",0,"",""],["the persistent data structure argument against linear consciousness","Tamsin Leake","2021","blog","carado.moe","carado.moe/persistent-data-structures-consciousness.html",0,"",""],["the systematic absence of libertarian thought","Tamsin Leake","2021","blog","carado.moe","carado.moe/systematic-unlibertarianism.html",0,"",""],["Vignettes workshop","Daniel Kokotajlo","2021","blog","aiimpacts.org","aiimpacts.org/vignettes-workshop/",0,"",""],["Vignettes Workshop (AI Impacts)","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jusSrXEAsiqehBsmh/vignettes-workshop-ai-impacts",0,"",""],["Vignettes Workshop (AI Impacts)","kokotajlod","2021","blog","EA Forum","forum.effectivealtruism.org/posts/6ajPou3jMjicwsnEs/vignettes-workshop-ai-impacts",0,"","forecasting"],["The case for strong longtermism","Hilary Greaves and William MacAskill","2021","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/hilary-greaves-william-macaskill-the-case-for-strong-longtermism-2/",0,"",""],["What is an example of recent, tangible progress in AI safety research?","Aaron Gertler","2021","blog","EA Forum","forum.effectivealtruism.org/posts/M5pGsPykoCnFgL6pS/what-is-an-example-of-recent-tangible-progress-in-ai-safety",0,"",""],["Answering questions honestly given world-model mismatches","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SRJ5J9Tnyq7bySxbt/answering-questions-honestly-given-world-model-mismatches",0,"",""],["Avoiding the instrumental policy by hiding information about humans","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/roZvoF6tRH6xYtHMF/avoiding-the-instrumental-policy-by-hiding-information-about",0,"","policy"],["Looking Deeper at Deconfusion","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5Nz4PJgvLCpJd6YTA/looking-deeper-at-deconfusion",0,"",""],["A New Formalism, Method and Open Issues for Zero-Shot Coordination","Johannes Treutlein and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.06613",0,"","agents"],["A naive alignment strategy and optimism about generalization","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QvtHSsZLFCAHmzes7/a-naive-alignment-strategy-and-optimism-about-generalization",0,"","policy robustness"],["Finite Factored Sets: Orthogonality and Time","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yT7QdN2wEubR8exAH/finite-factored-sets-orthogonality-and-time",0,"",""],["Hard Choices in Artificial Intelligence","Roel Dobbe and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.11022",0,"",""],["Knowledge is not just mutual information","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QLosiQsPJepZWtXG4/knowledge-is-not-just-mutual-information",0,"",""],["Synthesising Reinforcement Learning Policies through Set-Valued Inductive Rule Learning","Youri Coppens and 3 others","2021","paper","Trustworthy AI - Integrating Learning, Optimization and Reasoning\n  (2021), Lecture Notes in Computer Science, vol. 12641, pp. 163-179","arxiv.org/abs/2106.06009",0,"","interpretability benchmarks agents policy robustness"],["Humanities Research Ideas for Longtermists","Lizka","2021","blog","EA Forum","forum.effectivealtruism.org/posts/oTJ5vMNwdWiHj2iKL/humanities-research-ideas-for-longtermists",0,"","governance"],["PEBBLE: Feedback-Efficient Interactive Reinforcement Learning via Relabeling Experience and Unsupervised Pre-training","Kimin Lee and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.05091",0,"","rlhf agents policy"],["AXRP Episode 8 - Assistance Games with Dylan Hadfield-Menell","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fzFyCJ6gB9kBL9RqW/axrp-episode-8-assistance-games-with-dylan-hadfield-menell",0,"",""],["Big picture of phasic dopamine","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jrewt3rLFiKWrKuyZ/big-picture-of-phasic-dopamine",0,"",""],["Conservative Agency with Multiple Stakeholders","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gLfHp8XaWpfsmXyWZ/conservative-agency-with-multiple-stakeholders",0,"","agents"],["Curriculum Design for Teaching via Demonstrations: Theory and Applications","Gaurav Yengera and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.04696",0,"","agents policy"],["Dangerous optimisation includes variance minimisation","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nbvd4o9uDPe5whFxa/dangerous-optimisation-includes-variance-minimisation",0,"",""],["Definitions of intent suitable for algorithms","Hal Ashton","2021","paper","arXiv preprint","arxiv.org/abs/2106.04235",0,"","agents robustness"],["Engines of Power: Electricity, AI, and General-Purpose Military Transformations","Jeffrey Ding and Allan Dafoe","2021","paper","arXiv preprint","arxiv.org/abs/2106.04338",0,"",""],["Evan Hubinger on Homogeneity in Takeoff Speeds, Learned Optimization and Interpretability","Michaël Trazzi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/NFfZsWrzALPdw54NL/evan-hubinger-on-homogeneity-in-takeoff-speeds-learned",0,"","interpretability forecasting"],["Game-theoretic Alignment in terms of Attainable Utility","midco and TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/buaGz3aiqCotzjKie/game-theoretic-alignment-in-terms-of-attainable-utility",0,"",""],["Provably Robust Detection of Out-of-distribution Data (almost) for free","Alexander Meinke and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.04260",0,"","robustness"],["Supplement to \"Big picture of phasic dopamine\"","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BwaxYiJ3ZmXHLoZJ6/supplement-to-big-picture-of-phasic-dopamine",0,"",""],["Survey on AI existential risk scenarios","Sam Clarke and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/WiXePTj7KeEycbiwK/survey-on-ai-existential-risk-scenarios",0,"",""],["Survey on AI existential risk scenarios","Sam Clarke and 2 others","2021","blog","EA Forum","forum.effectivealtruism.org/posts/2tumunFmjBuXdfF2F/survey-on-ai-existential-risk-scenarios-1",0,"","governance forecasting"],["The reverse Goodhart problem","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/4RH5cMSBLZcv8DEw2/the-reverse-goodhart-problem",0,"","goodharts-law"],["The reverse Goodhart problem","Stuart_Armstrong","2021","blog","LessWrong","www.lesswrong.com/posts/4RH5cMSBLZcv8DEw2/the-reverse-goodhart-problem",0,"","goodharts-law"],["There Is No Turning Back: A Self-Supervised Approach for Reversibility-Aware Reinforcement Learning","Nathan Grinsztajn and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.04480",0,"","agents"],["Improving Social Welfare While Preserving Autonomy via a Pareto Mediator","Stephen McAleer and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.03927",0,"","agents"],["Moral Consideration of Nonhumans in the Ethics of Artificial Intelligence | Global Catastrophic Risk Institute","Seth Baum","2021","report","gcrinstitute.org","gcrinstitute.org/moral-consideration-of-nonhumans-in-the-ethics-of-artificial-intelligence/",0,"",""],["Some AI Governance Research Ideas","Alexis Carlier and markusanderljung","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/RBsTG5F2LqsMaqdzP/some-ai-governance-research-ideas",0,"","governance"],["Speculations against GPT-n writing alignment papers","Donald Hobson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wkhfytDQvfx3Jeie9/speculations-against-gpt-n-writing-alignment-papers",0,"","interpretability"],["Review of \"Learning Normativity: A Research Agenda\"","Gyrodiot and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ykvw6sMQD7JXK5cdJ/review-of-learning-normativity-a-research-agenda",0,"",""],["The dumbest kid in the world (joke)","CronoDAS","2021","blog","LessWrong","www.lesswrong.com/posts/cxDvhqDKn5W3eubvA/the-dumbest-kid-in-the-world-joke",0,"","theory"],["A Viral License for AI Safety","IvanVendrov","2021","blog","EA Forum","forum.effectivealtruism.org/posts/dsEMaqKNmArdCRGeH/a-viral-license-for-ai-safety",0,"","governance policy"],["High Impact Careers in Formal Verification: Artificial Intelligence","quinn","2021","blog","EA Forum","forum.effectivealtruism.org/posts/4rMxiyPTPdzaFMyGm/high-impact-careers-in-formal-verification-artificial",0,"",""],["Search-in-Territory vs Search-in-Map","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/s2KJWLAPyjtmQ9ze3/search-in-territory-vs-search-in-map",0,"",""],["The Nature of Counterfactuals","Chris_Leong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/T4Mef9ZkL4WftQBqw/the-nature-of-counterfactuals",0,"","theory"],["Finite Factored Sets: Introduction and Factorizations","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/sZa5LQg6rrWgMR4Jx/finite-factored-sets-introduction-and-factorizations",0,"",""],["I'm creating a world simulation video game","Tamsin Leake","2021","blog","carado.moe","carado.moe/game.html",0,"",""],["Reflection of Hierarchical Relationship via Nuanced Conditioning of Game Theory Approach for AI Development and Utilization","Kyoung-cheol Kim","2021","blog","LessWrong","www.lesswrong.com/posts/BCynDEwguEiogicAo/reflection-of-hierarchical-relationship-via-nuanced",0,"","agents governance"],["SIA is basically just Bayesian updating on existence","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BYy62ib5tAkn9rsKn/sia-is-basically-just-bayesian-updating-on-existence",0,"",""],["The underlying model of a morphism","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GRAWAqfgZEgtuCvje/the-underlying-model-of-a-morphism",0,"","deception"],["An Intuitive Guide to Garrabrant Induction","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/y5GftLezdozEHdXkL/an-intuitive-guide-to-garrabrant-induction",0,"","theory"],["Offline Reinforcement Learning as One Big Sequence Modeling Problem","Michael Janner and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.02039",0,"","benchmarks"],["Overcoming Narratives","Tamsin Leake","2021","blog","carado.moe","carado.moe/overcoming-narratives.html",0,"",""],["Rogue AGI Embodies Valuable Intellectual Property","Mark Xu and CarlShulman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FM49gHBrs5GTx7wFf/rogue-agi-embodies-valuable-intellectual-property",0,"",""],["Some AI Governance Research Ideas","MarkusAnderljung and ac","2021","blog","EA Forum","forum.effectivealtruism.org/posts/kvkv6779jk6edygug/some-ai-governance-research-ideas",0,"","governance policy compute-governance"],["Towards a Mathematical Theory of Abstraction","Beren Millidge","2021","paper","arXiv preprint","arxiv.org/abs/2106.01826",0,"",""],["Thoughts on the Alignment Implications of Scaling Language Models","leogao","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EmxfgPGvaKqhttPM8/thoughts-on-the-alignment-implications-of-scaling-language",0,"","scaling-laws"],["Why Release a Large Language Model?","Connor Leahy","2021","blog","blog.eleuther.ai","blog.eleuther.ai/why-release-a-large-language-model/",0,"",""],["\"Existential risk from AI\" survey results","Rob Bensinger","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QvwSr5LsxyDeaPK5s/existential-risk-from-ai-survey-results",0,"",""],["\"Existential risk from AI\" survey results","RobBensinger","2021","blog","EA Forum","forum.effectivealtruism.org/posts/8CM9vZ2nnQsWJNsHx/existential-risk-from-ai-survey-results",0,"","forecasting"],["CHAI Newsletter #1 2021","CHAI","2021","report","drive.google.com","drive.google.com/file/d/1IRHSuqsvPH4p0EudbBwoROGwXPBkoUSW/view?usp=sharing",0,"",""],["Final Report of the National Security Commission on Artificial Intelligence (NSCAI, 2021)","MichaelA","2021","blog","EA Forum","forum.effectivealtruism.org/posts/zwZLiKSgRwYRi9Jzt/final-report-of-the-national-security-commission-on",0,"","governance policy forecasting"],["Machine Learning and Cybersecurity","Micah Musser and Ashton Garriott","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/machine-learning-and-cybersecurity/",0,"",""],["Poison in the Well","Andrew Lohn","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/poison-in-the-well/",0,"",""],["What Matters for Adversarial Imitation Learning?","Manu Orsini and 9 others","2021","paper","arXiv preprint","arxiv.org/abs/2106.00672",0,"","benchmarks agents"],["How much will pre-transformative AI speed up R&D?","Ben Snodin","2021","blog","EA Forum","forum.effectivealtruism.org/posts/JNXAFnHbdQBMGDDxh/how-much-will-pre-transformative-ai-speed-up-r-and-d",0,"","forecasting"],["Institutionalising Ethics in AI through Broader Impact Requirements","Carina Prunkl and 5 others","2021","paper","Nature Machine Intelligence 3.2 (2021): 104-110","arxiv.org/abs/2106.11039",0,"","interpretability governance policy"],["[Event] Weekly Alignment Research Coffee Time","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cysgh8zpmvt56f6Qw/event-weekly-alignment-research-coffee-time",0,"",""],["AI Safety Career Bottlenecks Survey Responses Responses","Linda Linsefors","2021","blog","EA Forum","forum.effectivealtruism.org/posts/2pxGXYX2JrptvLpzZ/ai-safety-career-bottlenecks-survey-responses-responses",0,"",""],["AXRP Episode 7.5 - Forecasting Transformative AI from Biological Anchors with Ajeya Cotra","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CuDYhLLXq6FuHvGZc/axrp-episode-7-5-forecasting-transformative-ai-from",0,"","forecasting"],["Predict responses to the \"existential risk from AI\" survey","RobBensinger","2021","blog","EA Forum","forum.effectivealtruism.org/posts/iBTon2dRYwcoS9Jyr/predict-responses-to-the-existential-risk-from-ai-survey",0,"","forecasting"],["Teaching ML to answer questions honestly instead of predicting human answers","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QqwZ7cwEA2cxFEAun/teaching-ml-to-answer-questions-honestly-instead-of",0,"","policy"],["The blue-minimising robot and model splintering","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BeeirdrMXCPYZwgfj/the-blue-minimising-robot-and-model-splintering",0,"",""],["An Offline Risk-aware Policy Selection Method for Bayesian Markov Decision Processes","Giorgio Angelotti and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.13431",0,"","policy robustness"],["Interactive Explanations: Diagnosis and Repair of Reinforcement Learning Based Agent Behaviors","Christian Arzate Cruz and Takeo Igarashi","2021","paper","arXiv preprint","arxiv.org/abs/2105.12938",0,"","rlhf deception agents policy robustness"],["Long-Term Future Fund: May 2021 grant recommendations","abergal","2021","blog","EA Forum","forum.effectivealtruism.org/posts/diZWNmLRgcbuwmYn4/long-term-future-fund-may-2021-grant-recommendations",0,"","forecasting"],["List of good AI safety project ideas?","Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/stdfRDMF3sFpSsGeG/list-of-good-ai-safety-project-ideas",0,"","robustness"],["MDP models are determined by the agent architecture and the environmental dynamics","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/XkXL96H6GknCbT5QH/mdp-models-are-determined-by-the-agent-architecture-and-the",0,"","instrumental-convergence agents"],["Abstraction Talk","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/aNngRKJCyLZEBTZhy/abstraction-talk",0,"",""],["Decoupling deliberation from competition","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7jSvfeyh8ogu8GcE6/decoupling-deliberation-from-competition",0,"",""],["Knowledge is not just map/territory resemblance","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YLoXcquNkNdsteZYd/knowledge-is-not-just-map-territory-resemblance",0,"",""],["Activation Function Ablation","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/activation-fns/",0,"",""],["Controlling Intelligent Agents The Only Way We Know How: Ideal Bureaucratic Structure (IBS)","Justin Bullock","2021","blog","LessWrong","www.lesswrong.com/posts/iekoEYDLgC7efzbBv/controlling-intelligent-agents-the-only-way-we-know-how",0,"","agents governance"],["Evaluating Different Fewshot Description Prompts on GPT-3","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/prompts-gpt-fewshot/",0,"","evals"],["Finetuning Models on Downstream Tasks","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/tuning-on-eval-harness/",0,"",""],["On the Sizes of OpenAI API Models","Leo Gao","2021","blog","blog.eleuther.ai","blog.eleuther.ai/gpt3-model-sizes/",0,"",""],["Problems facing a correspondence theory of knowledge","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/YdxG2D3bvG5YsuHpG/problems-facing-a-correspondence-theory-of-knowledge",0,"",""],["True Few-Shot Learning with Language Models","Ethan Perez and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.11447",0,"","evals"],["Finite Factored Sets","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/N5Jm6Nj4HkNKySA5Z/finite-factored-sets",0,"",""],["Finite Factored Sets","Scott Garrabrant","2021","blog","intelligence.org","intelligence.org/2021/05/23/finite-factored-sets/",0,"",""],["[Event] Weekly Alignment Research Coffee Time (05/24)","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/KGJC6HLG5hcFR7pM4/event-weekly-alignment-research-coffee-time-05-24",0,"",""],["AI Safety Research Project Ideas","Owain_Evans","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/f69LK7CndhSNA7oPn/ai-safety-research-project-ideas",0,"",""],["Navigation Turing Test (NTT): Learning to Evaluate Human-Like Navigation","Sam Devlin and 8 others","2021","paper","Proceedings of the 38th International Conference on Machine\n  Learning (ICML), 139:2644-2653, 2021","arxiv.org/abs/2105.09637",0,"","rlhf evals agents"],["Response to \"What does the universal prior actually look like?\"","michaelcohen","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/n2Gseb3XFpMyc2FEb/response-to-what-does-the-universal-prior-actually-look-like",0,"",""],["[AN #151]: How sparsity in the final layer makes a neural net debuggable","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/t2yeWvpGvzQ9sFrWc/an-151-how-sparsity-in-the-final-layer-makes-a-neural-net",0,"",""],["A Tour of Emerging Cryptographic Technologies | GovAI","Ben Garfinkel","2021","report","governance.ai","www.governance.ai/research-paper/a-tour-of-emerging-cryptographic-technologies",0,"",""],["May 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/05/18/may-2021-newsletter/",0,"",""],["AI and Shared Prosperity","Katya Klinova and Anton Korinek","2021","paper","Proceedings of the 2021 AAAI/ACM Conference on AI, Ethics, and\n  Society (AIES '21)","arxiv.org/abs/2105.08475",0,"","policy"],["Modeling the Sequential Dependence among Audience Multi-step Conversions with Multi-task Learning in Targeted Display Advertising","Dongbo Xi and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.08489",0,"",""],["Saving Time","Scott Garrabrant","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gEKHX8WKrXGM4roRC/saving-time",0,"","theory"],["Saving Time","Scott Garrabrant","2021","blog","intelligence.org","intelligence.org/2021/05/18/saving-time/",0,"",""],["SGD's Bias","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ej2r2JADoWiEtxkCd/sgd-s-bias",0,"",""],["Unifying Principles and Metrics for Safe and Assistive AI","Siddharth Srivastava","2021","report","ojs.aaai.org","ojs.aaai.org/index.php/AAAI/article/view/17769",0,"",""],["Knowledge Neurons in Pretrained Transformers","evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LdoKzGom7gPLqEZyQ/knowledge-neurons-in-pretrained-transformers",0,"","interpretability"],["Why should we *not* put effort into AI safety research?","Ben Thompson","2021","blog","EA Forum","forum.effectivealtruism.org/posts/DKEoHRH2pvZxzBZN2/why-should-we-not-put-effort-into-ai-safety-research",0,"",""],["[Event] Weekly Alignment Research Coffee Time (05/17)","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gLRphsnSHefpcqZoF/event-weekly-alignment-research-coffee-time-05-17",0,"",""],["Intermittent Distillations #3","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jnHxfXgyQj3ALsD5a/intermittent-distillations-3",0,"",""],["Saving The Client-Side Web: just WASM and the DOM","Tamsin Leake","2021","blog","carado.moe","carado.moe/saving-the-web.html",0,"",""],["What harm could AI safety do?","SeanEngelhart","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ciKv8MRJ7gYyGS65o/what-harm-could-ai-safety-do",0,"",""],["Agree to Disagree: When Deep Learning Models With Identical Architectures Produce Distinct Explanations","Matthew Watson and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.06791",0,"","interpretability deception"],["AXRP Episode 7 - Side Effects with Victoria Krakovna","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/C9vj5ZX3KsgFfwXAN/axrp-episode-7-side-effects-with-victoria-krakovna",0,"","agents robustness"],["Our all-time largest donation, and major crypto support from Vitalik Buterin","Colm Ó Riain","2021","blog","intelligence.org","intelligence.org/2021/05/13/two-major-donations/",0,"",""],["Understanding the Lottery Ticket Hypothesis","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dpzLqQQSs7XRacEfK/understanding-the-lottery-ticket-hypothesis",0,"",""],["Agency in Conway’s Game of Life","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3SG4WbNPoP8fsuZgs/agency-in-conway-s-game-of-life",0,"",""],["[AN #150]: The subtypes of Cooperative AI research","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3zmKzbMPjPvEcZfkn/an-150-the-subtypes-of-cooperative-ai-research",0,"",""],["Formal Inner Alignment, Prospectus","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/a7jnbtoKFyvu5qfkd/formal-inner-alignment-prospectus",0,"",""],["Challenge: know everything that the best go bot knows about go","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/m5frrcYTSH6ENjsc9/challenge-know-everything-that-the-best-go-bot-knows-about",0,"","interpretability"],["Is driving worth the risk?","Adam Zerner","2021","blog","LessWrong","www.lesswrong.com/posts/AL6jdmpcxESxQTpfQ/is-driving-worth-the-risk",0,"","forecasting"],["Leveraging Sparse Linear Layers for Debuggable Deep Networks","Eric Wong and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.04857",0,"",""],["Yampolskiy on AI Risk Skepticism","Gordon Seidoh Worley","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/D3PnBxkj5jkKPm6jr/yampolskiy-on-ai-risk-skepticism",0,"",""],["Human priors, features and models, languages, and Solmonoff induction","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/9aFpMtpivqPCBfx2w/human-priors-features-and-models-languages-and-solmonoff",0,"",""],["[Event] Weekly Alignment Research Coffee Time (05/10)","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ErXseAhtiymqRdCq9/event-weekly-alignment-research-coffee-time-05-10",0,"",""],["Pre-Training + Fine-Tuning Favors Deception","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rZTjsKy4Jvu6krWJt/pre-training-fine-tuning-favors-deception",0,"","deception"],["Finding the unicorn: Predicting early stage startup success through a hybrid intelligence method","Dominik Dellermann and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.03360",0,"",""],["Life and expanding steerable consequences","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tmZRyXvH9dgopcnuE/life-and-expanding-steerable-consequences",0,"","forecasting"],["Using reinforcement learning to design an AI assistantfor a satisfying co-op experience","Ajay Krishnan and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.03414",0,"","evals agents"],["Adversarial Reprogramming of Neural Cellular Automata","Ettore Randazzo and 3 others","2021","report","Distill","distill.pub/selforg/2021/adversarial",0,"",""],["Anthropics: different probabilities, different questions","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LARmKTbpAkEYeG43u/anthropics-different-probabilities-different-questions",0,"",""],["Less Realistic Tales of Doom","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZRTr6rEcpYtfMTDBs/less-realistic-tales-of-doom",0,"",""],["Parsing Chris Mingard on Neural Networks","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5p4ynEJQ8nXxp2sxC/parsing-chris-mingard-on-neural-networks",0,"",""],["[AN #149]: The newsletter's editorial policy","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Yj9hW27sMJ4Hx4Bd4/an-149-the-newsletter-s-editorial-policy",0,"","policy"],["Ethics and Governance of Artificial Intelligence: Evidence from a Survey of Machine Learning Researchers","Baobao Zhang and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.02117",0,"","governance policy"],["Hard Choices and Hard Limits for Artificial Intelligence","Bryce Goodman","2021","paper","arXiv preprint","arxiv.org/abs/2105.07852",0,"",""],["Mundane solutions to exotic problems","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/d5m3G3ov5phZu7FX3/mundane-solutions-to-exotic-problems",0,"",""],["Mundane solutions to exotic problems","Paul Christiano","2021","report","ai-alignment.com","ai-alignment.com/mundane-solutions-to-exotic-problems-395bad49fbe7",0,"",""],["Parsing Abram on Gradations of Inner Alignment Obstacles","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pTm6aEvmepJEA5cuK/parsing-abram-on-gradations-of-inner-alignment-obstacles",0,"",""],["Consistencies as (meta-)preferences","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/KptP3J2ThDTnriric/consistencies-as-meta-preferences",0,"",""],["Hybrid Intelligence","Dominik Dellermann and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.00691",0,"",""],["RL-IoT: Reinforcement Learning to Interact with IoT Devices","Giulia Milan and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.00884",0,"",""],["[Weekly Event] Alignment Researcher Coffee Time (in Walled Garden)","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2Ps9easGbqdMP6win/weekly-event-alignment-researcher-coffee-time-in-walled",0,"",""],["AI Risk Skepticism","Roman V. Yampolskiy","2021","paper","arXiv preprint","arxiv.org/abs/2105.02704",0,"",""],["April 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/05/02/april-2021-newsletter/",0,"",""],["Planning for Proactive Assistance in Environments with Partial Observability","Anagha Kulkarni and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.00525",0,"","evals agents"],["pyBKT: An Accessible Python Library of Bayesian Knowledge Tracing Models","Anirudhan Badrinath and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.00385",0,"","evals"],["The Unsatisfactorily Far Reach Of Property","Tamsin Leake","2021","blog","carado.moe","carado.moe/unsatisfactory-property.html",0,"",""],["Video Games Needs A Platform","Tamsin Leake","2021","blog","carado.moe","carado.moe/video-games-needs-a-platform.html",0,"",""],["Contending Frames: Evaluating Rhetorical Dynamics in AI","Andrew Imbrie and 3 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/contending-frames/",0,"","evals"],["Cooperative AI: machines must learn to find common ground","Allan Dafoe and 5 others","2021","report","nature.com","www.nature.com/articles/d41586-021-01170-0",0,"",""],["Machine Intelligence for Scientific Discovery and Engineering Invention","Matthew Daniels and 3 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/machine-intelligence-for-scientific-discovery-and-engineering-invention/",0,"",""],["Symmetry, Equilibria, and Robustness in Common-Payoff Games","Scott Emmons and 4 others","2021","report","preflib.github.io","preflib.github.io/gaiw2021/papers/GAIW_2021_paper_32.pdf",0,"","robustness"],["The Societal Implications of Deep Reinforcement Learning","Jess Whittlestone and 2 others","2021","report","jair.org","jair.org/index.php/jair/article/view/12360",0,"",""],["Truth, Lies, and Automation","Ben Buchanan and 3 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/truth-lies-and-automation/",0,"",""],["Low-stakes alignment","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/TPan9sQFuPP6jgEJo/low-stakes-alignment",0,"",""],["Low-stakes alignment","Paul Christiano","2021","report","ai-alignment.com","ai-alignment.com/low-stakes-alignment-f3c36606937f",0,"","agents robustness"],["25 Min Talk on MetaEthical.AI with Questions from Stuart Armstrong","June Ku","2021","blog","LessWrong","www.lesswrong.com/posts/oAJ7Pd2PiBHT2cQ3p/25-min-talk-on-metaethical-ai-with-questions-from-stuart",0,"",""],["[AN #148]: Analyzing generalization across more axes than just accuracy or loss","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/H79dxa7XXMBhwqZLm/an-148-analyzing-generalization-across-more-axes-than-just",0,"",""],["AMA: Paul Christiano, alignment researcher","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7qhtuQLCCvmwCPfXK/ama-paul-christiano-alignment-researcher",0,"",""],["Draft report on existential risk from power-seeking AI","Joe Carlsmith","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HduCjmXTBD4xYTegv/draft-report-on-existential-risk-from-power-seeking-ai",0,"","instrumental-convergence power-seeking"],["Draft report on existential risk from power-seeking AI","Joe_Carlsmith","2021","blog","EA Forum","forum.effectivealtruism.org/posts/78NoGoRitPzeT8nga/draft-report-on-existential-risk-from-power-seeking-ai",0,"","power-seeking forecasting"],["Why AI is Harder Than We Think - Melanie Mitchell","BrownHairedEevee","2021","blog","EA Forum","forum.effectivealtruism.org/posts/C94JhsbSfZ8iPNedy/why-ai-is-harder-than-we-think-melanie-mitchell",0,"","forecasting"],["Agents Over Cartesian World Models","Mark Xu and evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LBNjeGaJZw7QdybMw/agents-over-cartesian-world-models",0,"","agents"],["Pitfalls of the agent model","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8HWGXhnCfAPgJYa9D/pitfalls-of-the-agent-model",0,"","agents policy"],["Plausible Quantum Suicide","Tamsin Leake","2021","blog","carado.moe","carado.moe/quantum-suicide.html",0,"",""],["[Linkpost] Treacherous turns in the wild","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/NEa3puQB23FyiifnW/linkpost-treacherous-turns-in-the-wild",0,"",""],["A new proposal for regulating AI in the EU","EdoArad","2021","blog","EA Forum","forum.effectivealtruism.org/posts/ARwvpA4dLvpPxNNRD/a-new-proposal-for-regulating-ai-in-the-eu",0,"","governance policy"],["Announcing the Alignment Research Center","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3ejHFgQihLG4L6WQf/announcing-the-alignment-research-center",0,"",""],["Axes for Sociotechnical Inquiry in AI Research","Sarah Dean and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2105.06551",0,"",""],["FAQ: Advice for AI Alignment Researchers","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/kdwk5aHNjM53PZFKL/faq-advice-for-ai-alignment-researchers",0,"",""],["Why AI is Harder Than We Think","Melanie Mitchell","2021","paper","arXiv preprint","arxiv.org/abs/2104.12871",0,"",""],["Beware over-use of the agent model","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2QuAcx8XQw7rrXzGC/beware-over-use-of-the-agent-model",0,"","agents"],["Causal Learning for Socially Responsible AI","Lu Cheng and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.12278",0,"",""],["Naturalism and AI alignment","Michele Campolo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Jo2LWuuGEGHHfGZCM/naturalism-and-ai-alignment",0,"",""],["Let's not generalize over people","Tamsin Leake","2021","blog","carado.moe","carado.moe/lets-not-generalize-politics.html",0,"",""],["Probability theory and logical induction as lenses","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zd5Bsra7ar2pa3bwS/probability-theory-and-logical-induction-as-lenses",0,"","theory"],["Is there anything that can stop AGI development in the near term?","Wulky Wilkinsen","2021","blog","LessWrong","www.lesswrong.com/posts/jCzZBgDkYYNqteH2j/is-there-anything-that-can-stop-agi-development-in-the-near",0,"","governance forecasting"],["NTK/GP Models of Neural Nets Can't Learn Features","interstice","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/76cReK4Mix3zKCWNT/ntk-gp-models-of-neural-nets-can-t-learn-features",0,"",""],["[AN #147]: An overview of the interpretability landscape","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zFwie6AoPyqGmMSsc/an-147-an-overview-of-the-interpretability-landscape",0,"","interpretability"],["Where are intentions to be found?","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EA4Txiuo5Ce2b7iBd/where-are-intentions-to-be-found",0,"",""],["Gradations of Inner Alignment Obstacles","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wpbpvjZCK3JhzpR2D/gradations-of-inner-alignment-obstacles",0,"",""],["Rotary Embeddings: A Relative Revolution","Stella Biderman and 7 others","2021","blog","blog.eleuther.ai","blog.eleuther.ai/rotary-embeddings/",0,"",""],["International cooperation as a tool to reduce two existential risks.","johl@umich.edu","2021","blog","EA Forum","forum.effectivealtruism.org/posts/fkN9zcqNeZGrXeeMF/international-cooperation-as-a-tool-to-reduce-two",0,"","governance policy"],["The Power of Scale for Parameter-Efficient Prompt Tuning","Brian Lester and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.08691",0,"","robustness"],["Updating the Lottery Ticket Hypothesis","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/i9p5KWNWcthccsxqm/updating-the-lottery-ticket-hypothesis",0,"",""],["Action Advising with Advice Imitation in Deep Reinforcement Learning","Ercument Ilhan and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.08441",0,"","agents policy"],["Learning on a Budget via Teacher Imitation","Ercument Ilhan and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.08440",0,"",""],["An EPIC way to evaluate reward functions","DeepMind Safety Research","2021","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/an-epic-way-to-evaluate-reward-functions-c2c6d41b61cc",0,"","evals"],["Superrational Agents Kelly Bet Influence!","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7EupfLrZ63pbdyb9J/superrational-agents-kelly-bet-influence",0,"","agents"],["Computing Natural Abstractions: Linear Approximation","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/f6oWbqxEwktfPrKJw/computing-natural-abstractions-linear-approximation",0,"",""],["Gradient-based Adversarial Attacks against Text Transformers","Chuan Guo and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.13733",0,"","robustness"],["[AN #146]: Plausible stories of how we might fail to avert an existential catastrophe","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AwxBGFy59DYDk4ooe/an-146-plausible-stories-of-how-we-might-fail-to-avert-an",0,"",""],["An Interpretability Illusion for BERT","Tolga Bolukbasi and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.07143",0,"","interpretability"],["Detection of Dataset Shifts in Learning-Enabled Cyber-Physical Systems using Variational Autoencoder for Regression","Feiyang Cai and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.06613",0,"","evals robustness monitoring"],["Intermittent Distillations #2","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Rjrq6xPoavgC4JznB/intermittent-distillations-2",0,"",""],["What if AGI is near?","Wulky Wilkinsen","2021","blog","LessWrong","www.lesswrong.com/posts/FQqXxWHyZ5AaYiZvt/what-if-agi-is-near",0,"","forecasting"],["Fiction relevant to AI futurism","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/partially-plausible-fictional-ai-futures/",0,"",""],["Is there evidence that recommender systems are changing users' preferences?","zdgroff","2021","blog","EA Forum","forum.effectivealtruism.org/posts/CHfuH58thMHPN8zHX/is-there-evidence-that-recommender-systems-are-changing",0,"","governance"],["The Atari Data Scraper","Brittany Davis Pierson and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.04893",0,"","agents"],["Working in Congress (Part #1): Background and some EA cause area analysis","US Policy Careers","2021","blog","EA Forum","forum.effectivealtruism.org/posts/otQtErQEB6R4GCDwF/working-in-congress-part-1-background-and-some-ea-cause-area-1",0,"","evals governance policy"],["Adapting Language Models for Zero-shot Learning by Meta-tuning on Dataset and Prompt Collections","Ruiqi Zhong and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.04670",0,"","evals deception"],["[AN #145]: Our three year anniversary!","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bER8yqrmHatrES9nR/an-145-our-three-year-anniversary",0,"",""],["A Framework for Ethical AI at the United Nations","Lambert Hogenhout","2021","paper","arXiv preprint","arxiv.org/abs/2104.12547",0,"","governance policy"],["Identifiability Problem for Superrational Decision Theories","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/iNGXKB8iExpcLvu55/identifiability-problem-for-superrational-decision-theories",0,"","theory"],["My Current Take on Counterfactuals","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXfka98pZXAmXiyDp/my-current-take-on-counterfactuals",0,"","theory"],["Opinions on Interpretable Machine Learning and 70 Summaries of Recent Papers","Peter Hase","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GEPX7jgLMB8vR2qaK/opinions-on-interpretable-machine-learning-and-70-summaries",0,"","interpretability evals training-data"],["Why unriggable *almost* implies uninfluenceable","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/k3J3sYgmjMmpkzbbc/why-unriggable-almost-implies-uninfluenceable",0,"",""],["A possible preference algorithm","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2SrzejmaxnwJBNkFE/a-possible-preference-algorithm",0,"",""],["AXRP Episode 6 - Debate and Imitative Generalization with Beth Barnes","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/behyPgMWFhXpKi73P/axrp-episode-6-debate-and-imitative-generalization-with-beth",0,"","scalable-oversight debate"],["Could Advanced AI Drive Explosive Economic Growth?","Tom Davidson","2021","report","openphilanthropy.org","www.openphilanthropy.org/could-advanced-ai-drive-explosive-economic-growth",0,"",""],["If you don't design for extrapolation, you'll extrapolate poorly - possibly fatally","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/4wa9XGnJHB3apPqoq/if-you-don-t-design-for-extrapolation-you-ll-extrapolate",0,"",""],["Learning What To Do by Simulating the Past","David Lindner and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.03946",0,"","rlhf agents policy"],["Solving the whole AGI control problem, version 0.0001","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gfw7JMdKirxeSPiAk/solving-the-whole-agi-control-problem-version-0-0001",0,"","interpretability"],["Voluntary safety commitments provide an escape from over-regulation in AI development","The Anh Han and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.03741",0,"","governance"],["Weight Banding","Michael Petrov and 5 others","2021","report","Distill","distill.pub/2020/circuits/weight-banding",0,"",""],["What do coherence arguments imply about the behavior of advanced AI?","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/what-do-coherence-arguments-imply-about-the-behavior-of-advanced-ai/",0,"",""],["Alignment Newsletter Three Year Retrospective","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/L7yHdqRiHKd3FhQ7B/alignment-newsletter-three-year-retrospective",0,"","rlhf evals robustness"],["Another (outer) alignment failure story","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AyNHoTWWAJ5eb99ji/another-outer-alignment-failure-story",0,"","evals agents robustness"],["Scaling Scaling Laws with Board Games","Andy L. Jones","2021","paper","arXiv preprint","arxiv.org/abs/2104.03113",0,"","policy forecasting scaling-laws"],["Which counterfactuals should an AI follow?","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/j7kyt6sHEjukRND8B/which-counterfactuals-should-an-ai-follow",0,"",""],["Case studies of self-governance to reduce technology risk","jia","2021","blog","EA Forum","forum.effectivealtruism.org/posts/Xf6QE6txgvfCGvZpk/case-studies-of-self-governance-to-reduce-technology-risk",0,"","governance"],["Coherence arguments imply a force for goal-directed behavior","Katja_Grace","2021","blog","EA Forum","forum.effectivealtruism.org/posts/wBbdQpy6dCjnMgxpJ/coherence-arguments-imply-a-force-for-goal-directed-behavior",0,"",""],["Reflective Bayesianism","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vpvLqinp4FoigqvKy/reflective-bayesianism",0,"",""],["Testing The Natural Abstraction Hypothesis: Project Intro","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cy3BhHrGinZCp3LXE/testing-the-natural-abstraction-hypothesis-project-intro",0,"","robustness"],["The Many Faces of Infra-Beliefs","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GS5P7LLLbSSExb3Sk/the-many-faces-of-infra-beliefs",0,"","theory"],["Branch Specialization","Chelsea Voss and 5 others","2021","report","Distill","distill.pub/2020/circuits/branch-specialization",0,"",""],["Risk Budgets vs. Basic Decision Theory","Vlad Firoiu","2021","blog","LessWrong","www.lesswrong.com/posts/BBvcAPDM9u6bYMMqi/risk-budgets-vs-basic-decision-theory",0,"","theory"],["How do scaling laws work for fine-tuning?","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/2j7mtf58Zr9XehjxP/how-do-scaling-laws-work-for-fine-tuning",0,"","scaling-laws"],["\"AI and Compute\" trend isn't predictive of what is happening","alexlyzhov","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wfpdejMWog4vEDLDg/ai-and-compute-trend-isn-t-predictive-of-what-is-happening",0,"",""],["[AN #144]: How language models can also be finetuned for non-language tasks","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/by5NkEoSC4gvo9bQ2/an-144-how-language-models-can-also-be-finetuned-for-non",0,"",""],["Artificial intelligence, human rights, democracy, and the rule of law: a primer","David Leslie and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.04147",0,"","policy"],["GPT-3 on Coherent Extrapolated Volition","janus","2021","blog","generative.ink","generative.ink/posts/gpt-3-on-coherent-extrapolated-volition/",0,"",""],["Learning Russian Roulette","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/L5Tf34FXA6weiGwEz/learning-russian-roulette",0,"","theory"],["My take on Michael Littman on \"The HCI of HAI\"","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wydAtj6FkPDHkdtzS/my-take-on-michael-littman-on-the-hci-of-hai",0,"",""],["Phylactery Decision Theory","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pba68kdmmHrp9oHGG/phylactery-decision-theory",0,"","theory"],["April files","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/april-drafts/",0,"",""],["Ethics and Artificial Intelligence","Jamie Baker","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ethics-and-artificial-intelligence/",0,"",""],["Formal Methods for the Informal Engineer: Workshop Recommendations","Gopal Sarma and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.00739",0,"",""],["March 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/04/01/march-2021-newsletter/",0,"",""],["DEALIO: Data-Efficient Adversarial Learning for Imitation from Observation","Faraz Torabi and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2104.00163",0,"","agents policy"],["Value and Earning","Tamsin Leake","2021","blog","carado.moe","carado.moe/value.html",0,"",""],["What Multipolar Failure Looks Like, and Robust Agent-Agnostic Processes (RAAPs)","Andrew_Critch","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LpM3EAakwYdS6aRKf/what-multipolar-failure-looks-like-and-robust-agent-agnostic",0,"","agents"],["Alignment of Language Agents","DeepMind Safety Research","2021","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/alignment-of-language-agents-9fbc7dd52c6c",0,"","agents"],["How do we prepare for final crunch time?","Eli Tyre","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wyYubb3eC5FS365nk/how-do-we-prepare-for-final-crunch-time",0,"",""],["Optimization, speculations on the X and only X problem.","Donald Hobson","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PhKZgz5Gxw9soHtng/optimization-speculations-on-the-x-and-only-x-problem",0,"",""],["\"Weak AI\" is Likely to Never Become \"Strong AI\", So What is its Greatest Value for us?","Bin Liu","2021","paper","arXiv preprint","arxiv.org/abs/2103.15294",0,"",""],["Towards An Ethics-Audit Bot","Siani Pearson and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.15746",0,"","governance"],["A Bayesian Approach to Identifying Representational Errors","Ramya Ramakrishnan and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.15171",0,"","evals agents policy"],["Cultural and Memetic Hygiene","Tamsin Leake","2021","blog","carado.moe","carado.moe/cultural-and-memetic-hygiene.html",0,"",""],["Infra-Domain proofs 1","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/H5zo4L7yv4bnBgexQ/infra-domain-proofs-1",0,"",""],["Infra-Domain Proofs 2","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/fLRgddjMTBnpbMeiM/infra-domain-proofs-2",0,"",""],["Inframeasures and Domain Theory","Diffractor","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/vrbidMiczaoHBhZGp/inframeasures-and-domain-theory",0,"",""],["Review of \"Fun with +12 OOMs of Compute\"","adamShimi and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bAAtiG8og7CxH3cXG/review-of-fun-with-12-ooms-of-compute",0,"","forecasting"],["Transparency Trichotomy","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cgJ447adbMAeoKTSt/transparency-trichotomy",0,"","interpretability"],["Alignment of Language Agents","Zachary Kenton and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.14659",0,"","deception agents"],["Coherence arguments imply a force for goal-directed behavior","KatjaGrace","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DkcdXsP56g9kXyBdq/coherence-arguments-imply-a-force-for-goal-directed-behavior",0,"","instrumental-convergence"],["Coherence arguments imply a force for goal-directed behavior","Katja Grace","2021","blog","aiimpacts.org","aiimpacts.org/coherence-arguments-imply-a-force-for-goal-directed-behavior/",0,"",""],["Report on Semi-informative Priors for AI timelines (Open Philanthropy)","Tom_Davidson","2021","blog","EA Forum","forum.effectivealtruism.org/posts/FPXFtBQHhkDDHBSt6/report-on-semi-informative-priors-for-ai-timelines-open",0,"","governance forecasting"],["Characterizing and Detecting Mismatch in Machine-Learning-Enabled Systems","Grace A. Lewis and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.14101",0,"",""],["My AGI Threat Model: Misaligned Model-Based RL Agent","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/zzXawbXDwCZobwF9D/my-agi-threat-model-misaligned-model-based-rl-agent",0,"","specification-gaming goodharts-law agents"],["Why 1-boxing doesn't imply backwards causation","Chris_Leong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/gAAFzqJkfeSHvcwTw/why-1-boxing-doesn-t-imply-backwards-causation",0,"","theory"],["[AN #143]: How to make embedded agents that reason probabilistically about their environments","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/beLgLr6edbZw4koh2/an-143-how-to-make-embedded-agents-that-reason",0,"","agents"],["Counterfactual Explanation with Multi-Agent Reinforcement Learning for Drug Target Prediction","Tri Minh Nguyen and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.12983",0,"","interpretability benchmarks deception agents"],["Toward Building Science Discovery Machines","Abdullah Khalili and Abdelhamid Bouchachia","2021","paper","arXiv preprint","arxiv.org/abs/2103.15551",0,"",""],["Toy model of preference, bias, and extra information","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/DQhwrir3nCcMtqA2j/toy-model-of-preference-bias-and-extra-information",0,"",""],["W2WNet: a two-module probabilistic Convolutional Neural Network with embedded data cleansing functionality","Francesco Ponzio and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.13107",0,"","benchmarks"],["Against evolution as an analogy for how humans will create AGI","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pz7Mxyr7Ac43tWMaC/against-evolution-as-an-analogy-for-how-humans-will-create",0,"",""],["AGI risk: analogies & arguments","Gavin","2021","blog","EA Forum","forum.effectivealtruism.org/posts/jeybxkZrJmWpJaatN/agi-risk-analogies-and-arguments",0,"",""],["Assured Learning-enabled Autonomy: A Metacognitive Reinforcement Learning Framework","Aquib Mustafa and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.12558",0,"","evals agents policy"],["Preferences and biases, the information argument","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dh8WsHfzmQJ5L7bd8/preferences-and-biases-the-information-argument",0,"",""],["Replacing Rewards with Examples: Example-Based Policy Search via Recursive Classification","Benjamin Eysenbach and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.12656",0,"","policy"],["Bridging Offline Reinforcement Learning and Imitation Learning: A Tale of Pessimism","Paria Rashidinejad and 4 others","2021","paper","Published at NeurIPS 2021 and IEEE Transactions on Information\n  Theory","arxiv.org/abs/2103.12021",0,"","policy"],["Combining Reward Information from Multiple Sources","Dmitrii Krasheninnikov and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.12142",0,"","deception agents robustness"],["Generalizing POWER to multi-agent games","midco and TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MJc9AqyMWpG3BqfyK/generalizing-power-to-multi-agent-games",0,"","instrumental-convergence agents"],["My research methodology","paulfchristiano","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EF5M6CmKRd6qZk27Z/my-research-methodology",0,"","scalable-oversight rlhf specification-gaming robustness"],["Fisherian Runaway as a decision-theoretic problem","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PvbzCuj293D5Hxvu3/fisherian-runaway-as-a-decision-theoretic-problem",0,"",""],["Introducing The Nonlinear Fund: AI Safety research, incubation, and funding","Kat Woods","2021","blog","EA Forum","forum.effectivealtruism.org/posts/fX8JsabQyRSd7zWiD/introducing-the-nonlinear-fund-ai-safety-research-incubation",0,"",""],["[AN #142]: The quest to understand a network well enough to reimplement it by hand","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/JGByt8TrxREo4twaw/an-142-the-quest-to-understand-a-network-well-enough-to",0,"",""],["HCH Speculation Post #2A","Charlie Steiner","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MnCMkh7hirX8YwT2t/hch-speculation-post-2a",0,"",""],["Intermittent Distillations #1","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/pqkdsqd6s6w2HtT9g/intermittent-distillations-1",0,"",""],["AI Policy Levers: A Review of the U.S. Government’s Tools to Shape AI Research, Development, and Deployment | GovAI","Sophie-Charlotte Fischer and 7 others","2021","report","governance.ai","www.governance.ai/research-paper/ai-policy-levers-a-review-of-the-u-s-governments-tools-to-shape-ai-research-development-and-deployment",0,"","policy"],["Comments on \"The Singularity is Nowhere Near\"","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/P7P2iG4zvBNANvQFK/comments-on-the-singularity-is-nowhere-near",0,"","forecasting"],["Lyapunov Barrier Policy Optimization","Harshit Sikchi and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.09230",0,"","agents policy"],["Using AI ethically to tackle covid-19","Stephen Cave and 4 others","2021","report","bmj.com","www.bmj.com/content/372/bmj.n364",0,"",""],["AI x-risk reduction: why I chose academia over industry","David Scott Krueger (formerly: capybaralet)","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/4jFnquoHuoaTqdphu/ai-x-risk-reduction-why-i-chose-academia-over-industry",0,"",""],["Success Weighted by Completion Time: A Dynamics-Aware Evaluation Criteria for Embodied Navigation","Naoki Yokoyama and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.08022",0,"","evals agents"],["Is Democracy a Fad?","bgarfinkel","2021","blog","EA Forum","forum.effectivealtruism.org/posts/TMCWXTayji7gvRK9p/is-democracy-a-fad",0,"","governance policy"],["\"Beliefs\" vs. \"Notions\"","David Scott Krueger (formerly: capybaralet)","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CmmhFtCg7hsAy3brQ/beliefs-vs-notions",0,"",""],["Towards Risk Modeling for Collaborative AI","Matteo Camilli and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.07460",0,"","assurance governance"],["Behavioral Sufficient Statistics for Goal-Directedness","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jkRFZNAZmWskTdCSt/behavioral-sufficient-statistics-for-goal-directedness",0,"",""],["Four Motivations for Learning Normativity","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/oqghwKKifztYWLsea/four-motivations-for-learning-normativity",0,"",""],["Resolutions to the Challenge of Resolving Forecasts","Davidmanheim","2021","blog","LessWrong","www.lesswrong.com/posts/JnDEAmNhSpBRpjD8L/resolutions-to-the-challenge-of-resolving-forecasts",0,"","goodharts-law forecasting"],["Symbolic Reinforcement Learning for Safe RAN Control","Alexandros Nikou and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.06602",0,"","agents"],["Systematic Mapping Study on the Machine Learning Lifecycle","Yuanhao Xie and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.10248",0,"",""],["TASP Ep 3 - Optimal Policies Tend to Seek Power","Quinn","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/eM6SgkXDbFXav4kD4/tasp-ep-3-optimal-policies-tend-to-seek-power",0,"","instrumental-convergence"],["[AN #141]: The case for practicing alignment work on GPT-3 and other large models","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/29QmG4bQDFtAzSmpv/an-141-the-case-for-practicing-alignment-work-on-gpt-3-and",0,"",""],["AXRP Episode 5 - Infra-Bayesianism with Vanessa Kosoy","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/FkMPXiomjGBjMfosg/axrp-episode-5-infra-bayesianism-with-vanessa-kosoy",0,"","theory"],["Designing Disaggregated Evaluations of AI Systems: Choices, Considerations, and Tradeoffs","Solon Barocas and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.06076",0,"","evals deception"],["Extended Picture Theory or Models inside Models inside Models","Chris_Leong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nvLNjY7aoh2i7JxbB/extended-picture-theory-or-models-inside-models-inside",0,"",""],["Open Problems with Myopia","Mark Xu and evhub","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LCLBnmwdxkkz5fNvH/open-problems-with-myopia",0,"","theory"],["[Link post] Coordination challenges for preventing AI conflict","stefan.torges","2021","blog","EA Forum","forum.effectivealtruism.org/posts/Sz4myiNgqmHjr2MA7/link-post-coordination-challenges-for-preventing-ai-conflict",0,"","governance"],["CLR's recent work on multi-agent systems","JesseClifton","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EzoCZjTdWTMgacKGS/clr-s-recent-work-on-multi-agent-systems",0,"","agents"],["Coordination challenges for preventing AI conflict","Stefan Torges","2021","report","longtermrisk.org","longtermrisk.org/coordination-challenges-for-preventing-ai-conflict/",0,"",""],["Pretrained Transformers as Universal Computation Engines","Kevin Lu and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.05247",0,"","training-data"],["The AI Index 2021 Annual Report","Daniel Zhang and 12 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.06312",0,"","policy"],["Towards a Mechanistic Understanding of Goal-Directedness","Mark Xu","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nTiAyxFybZ7jgtWvn/towards-a-mechanistic-understanding-of-goal-directedness",0,"",""],["A simple way to make GPT-3 follow instructions","Quintin Pope","2021","blog","LessWrong","www.lesswrong.com/posts/2H4huFGykKCP5Qu7C/a-simple-way-to-make-gpt-3-follow-instructions",0,"",""],["Epistemological Framing for AI Alignment Research","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y4YHTBziAscS5WPN7/epistemological-framing-for-ai-alignment-research",0,"",""],["Multi-agent learning in mixed-motive coordination problems","Julian Stastny and 3 others","2021","report","longtermrisk.org","longtermrisk.org/files/stastny_et_al_implicit_bargaining.pdf",0,"","agents"],["What I'd change about different philosophy fields","Rob Bensinger","2021","blog","LessWrong","www.lesswrong.com/posts/3Lyki5DCHnJgeNXww/what-i-d-change-about-different-philosophy-fields",0,"","theory"],["Collaborative game specification: arriving at common models in bargaining","Jesse Clifton","2021","report","longtermrisk.org","longtermrisk.org/collaborative-game-specification/",0,"","deception agents"],["Causal Analysis of Agent Behavior for AI Safety","Grégoire Déletang and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.03938",0,"","agents"],["MIRI comments on Cotra's \"Case for Aligning Narrowly Superhuman Models\"","Rob Bensinger","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AyfDnnAdjG7HHeD3d/miri-comments-on-cotra-s-case-for-aligning-narrowly",0,"","interpretability"],["Multimodal Neurons in Artificial Neural Networks","Kaj_Sotala","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/bgQysKL6Luqacw3SN/multimodal-neurons-in-artificial-neural-networks",0,"",""],["Rissanen Data Analysis: Examining Dataset Characteristics via Description Length","Ethan Perez and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.03872",0,"","evals"],["The case for aligning narrowly superhuman models","Ajeya Cotra","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/PZtsoaoSLpKjjbMqM/the-case-for-aligning-narrowly-superhuman-models",0,"","rlhf robustness training-data"],["What mechanisms drive agent behaviour?","DeepMind Safety Research","2021","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/what-mechanisms-drive-agent-behaviour-e7b8d9aee88",0,"","agents"],["[AN #140]: Theoretical models that predict scaling laws","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Yt5wAXMc7D2zLpQqx/an-140-theoretical-models-that-predict-scaling-laws",0,"","scaling-laws"],["A non-logarithmic argument for Kelly","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HLCcTypehEJtstNnD/a-non-logarithmic-argument-for-kelly",0,"","theory"],["A Semitechnical Introductory Dialogue on Solomonoff Induction","Eliezer Yudkowsky","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EL4HNa92Z95FKL9R2/a-semitechnical-introductory-dialogue-on-solomonoff-1",0,"","robustness"],["Book review: \"A Thousand Brains\" by Jeff Hawkins","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ixZLTmFfnKRbaStA5/book-review-a-thousand-brains-by-jeff-hawkins",0,"",""],["Connecting the good regulator theorem with semantics and symbol grounding","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/uky9nAtnw9WrAjziD/connecting-the-good-regulator-theorem-with-semantics-and",0,"","robustness"],["From-above vs Fine-grain diversity","Tamsin Leake","2021","blog","carado.moe","carado.moe/from-above-fine-grain-diversity.html",0,"",""],["Growth Doesn't Care About Crises","Tamsin Leake","2021","blog","carado.moe","carado.moe/growth-doesnt-care-about-crises.html",0,"",""],["How does bee learning compare with machine learning?","eleni","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yW3Tct2iyBMzYhTw7/how-does-bee-learning-compare-with-machine-learning",0,"","forecasting"],["Normies Are in Hell Too","Tamsin Leake","2021","blog","carado.moe","carado.moe/normies-are-in-hell-too.html",0,"",""],["Symbology for Topia","Tamsin Leake","2021","blog","carado.moe","carado.moe/symbology-for-topia.html",0,"",""],["Value Crystallization","Tamsin Leake","2021","blog","carado.moe","carado.moe/value-crystallization.html",0,"",""],["Evaluating Robustness of Counterfactual Explanations","André Artelt and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.02354",0,"","interpretability evals robustness"],["The Importance of Artificial Sentience","Jamie_Harris","2021","blog","EA Forum","forum.effectivealtruism.org/posts/cEqBEeNrhKzDp25fH/the-importance-of-artificial-sentience",0,"","governance"],["February 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/03/02/february-2021-newsletter/",0,"",""],["AI, Governance Displacement, and the (De)Fragmentation of International Law","Matthijs M. Maas","2021","report","ssrn.com","www.ssrn.com/abstract=3806624",0,"","governance"],["Full-time AGI Safety!","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/tnEQMnpyBFK5QBRz3/full-time-agi-safety",0,"",""],["Fun with +12 OOMs of Compute","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rzqACeBGycZtqCfaX/fun-with-12-ooms-of-compute",0,"","forecasting"],["International Control of Powerful Technology: Lessons from the Baruch Plan for Nuclear Weapons","Waqar Zaidi and Allan Dafoe","2021","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/2021/03/International-Control-of-Powerful-Technology-Lessons-from-the-Baruch-Plan-Zaidi-Dafoe-2021.pdf",0,"","governance policy"],["Key Concepts in AI Safety: An Overview","Tim G. J. Rudner and Helen Toner","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/key-concepts-in-ai-safety-an-overview/",0,"",""],["Key Concepts in AI Safety: Interpretability in Machine Learning","Tim G. J. Rudner and Helen Toner","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/key-concepts-in-ai-safety-interpretability-in-machine-learning/",0,"","interpretability"],["Key Concepts in AI Safety: Robustness and Adversarial Examples","Tim G. J. Rudner and Helen Toner","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/key-concepts-in-ai-safety-robustness-and-adversarial-examples/",0,"","robustness"],["How might cryptocurrencies affect AGI timelines?","Dawn Drescher","2021","blog","LessWrong","www.lesswrong.com/posts/NQweRxjPTyLZNQWKB/how-might-cryptocurrencies-affect-agi-timelines",0,"","forecasting"],["Bootstrapped Alignment","Gordon Seidoh Worley","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/teCsd4Aqg9KDxkaC9/bootstrapped-alignment",0,"","goodharts-law"],["List sorting does not play well with few-shot","janus","2021","blog","generative.ink","generative.ink/posts/list-sorting-does-not-play-well-with-few-shot/",0,"",""],["Secure Evaluation of Knowledge Graph Merging Gain","Leandro Eichenberger and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2103.00082",0,"","evals robustness"],["Bias-reduced Multi-step Hindsight Experience Replay for Efficient Multi-goal Reinforcement Learning","Rui Yang and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.12962",0,"","policy"],["Google's ethics is alarming","len.hoang.lnh","2021","blog","EA Forum","forum.effectivealtruism.org/posts/Zncu6QpJLJRSGofvK/google-s-ethics-is-alarming",0,"","governance"],["Is there any serious attempt to create a system to figure out the CEV of humanity and if not, why haven't we started yet?","Jonas Hallgren","2021","blog","LessWrong","www.lesswrong.com/posts/wTW4Juw49rHwAnxQh/is-there-any-serious-attempt-to-create-a-system-to-figure",0,"",""],["[AN #139]: How the simplicity of reality explains the success of neural nets","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6kgBAJBGp5Yum8oGj/an-139-how-the-simplicity-of-reality-explains-the-success-of",0,"",""],["Beyond Fine-Tuning: Transferring Behavior in Reinforcement Learning","Víctor Campos and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.13515",0,"","agents"],["Zero-Shot Text-to-Image Generation","Aditya Ramesh and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.12092",0,"","evals"],["[Fiction] Lena (MMAcevedo)","Kaj_Sotala","2021","blog","LessWrong","www.lesswrong.com/posts/yWMKQBnTwFAPFdN6S/fiction-lena-mmacevedo",0,"",""],["A Citizen's Guide to Artificial Intelligence","John Zerilli","2021","report","goodreads.com","www.goodreads.com/book/show/53730358-a-citizen-s-guide-to-artificial-intelligence",0,"",""],["Software Architecture for Next-Generation AI Planning Systems","Sebastian Graef and Ilche Georgievski","2021","paper","arXiv preprint","arxiv.org/abs/2102.10985",0,"","benchmarks"],["A Game-Theoretic Approach for Hierarchical Epidemic Control","Feiran Jia and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.10646",0,"","policy"],["Interview with Tom Chivers: “AI is a plausible existential risk, but it feels as if I’m in Pascal’s mugging”","felix.h","2021","blog","EA Forum","forum.effectivealtruism.org/posts/feNJWCo4LbsoKbRon/interview-with-tom-chivers-ai-is-a-plausible-existential",0,"",""],["Google’s Ethical AI team and AI Safety","magfrump","2021","blog","LessWrong","www.lesswrong.com/posts/WNkcqhAQPPiDjSDaB/google-s-ethical-ai-team-and-ai-safety-1",0,"","governance"],["How my school gamed the stats","Srdjan Miletic","2021","blog","LessWrong","www.lesswrong.com/posts/Yv9aj9bWD5H7aaDdy/how-my-school-gamed-the-stats",0,"","goodharts-law"],["A maximum entropy model of bounded rational decision-making with prior beliefs and market feedback","Benjamin Patrick Evans and Mikhail Prokopenko","2021","paper","arXiv preprint","arxiv.org/abs/2102.09180",0,"","agents"],["AXRP Episode 4 - Risks from Learned Optimization with Evan Hubinger","DanielFilan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/EszCTbovFfpJd5C8N/axrp-episode-4-risks-from-learned-optimization-with-evan",0,"",""],["Formal Solution to the Inner Alignment Problem","michaelcohen","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CnruhwFGQBThvgJiX/formal-solution-to-the-inner-alignment-problem",0,"","deception agents"],["Training a Resilient Q-Network against Observational Interference","Chao-Han Huck Yang and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.09677",0,"","evals benchmarks agents"],["Utility Maximization = Description Length Minimization","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/voLHQgNncnjjgAPH7/utility-maximization-description-length-minimization",0,"",""],["2021-03-01 National Library of Medicine Presentation: “Atlas of AI: Mapping the social and economic forces behind AI”","IrenicTruth","2021","blog","LessWrong","www.lesswrong.com/posts/x46D2DdxmRjFfsnBh/2021-03-01-national-library-of-medicine-presentation-atlas",0,"","governance"],["[AN #138]: Why AI governance should find problems rather than just solving them","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/XJqtRWnNRLaqJ8RCx/an-138-why-ai-governance-should-find-problems-rather-than",0,"","governance"],["Fully General Online Imitation Learning","Michael K. Cohen and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.08686",0,"","robustness"],["Graphical World Models, Counterfactuals, and Machine Learning Agents","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/q4j7qbEZRaTAA9Kxf/graphical-world-models-counterfactuals-and-machine-learning",0,"","agents theory"],["Safely controlling the AGI agent reward function","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/o3smzgcH8MR9RcMgZ/safely-controlling-the-agi-agent-reward-function",0,"","reward-hacking agents"],["Cartesian frames as generalised models","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wiQeYuQPwSypXXFar/cartesian-frames-as-generalised-models",0,"",""],["Disentangling Corrigibility: 2015-2021","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/MiYkTp6QYKXdJbchu/disentangling-corrigibility-2015-2021",0,"","reward-hacking"],["Generalised models as a category","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nQxqSsHfexivsd6vB/generalised-models-as-a-category",0,"",""],["Mathematical Models of Progress?","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ysQEJ8tvm8KYc76D5/mathematical-models-of-progress",0,"","forecasting"],["Suggestions of posts on the AF to review","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/6hdxTTPWF2iAbXjAb/suggestions-of-posts-on-the-af-to-review",0,"",""],["Transferring Domain Knowledge with an Adviser in Continuous Tasks","Rukshan Wijesinghe and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.08029",0,"","benchmarks agents policy"],["How RL Agents Behave When Their Actions Are Modified","Eric D. Langlois and Tom Everitt","2021","paper","Proceedings of the AAAI Conference on Artificial Intelligence,\n  35(13), 11586-11594 (2021)","arxiv.org/abs/2102.07716",0,"","agents policy"],["Machine Learning Model Development from a Software Engineering Perspective: A Systematic Literature Review","Giuliano Lorenzoni and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.07574",0,"","deception"],["On the Equilibrium Elicitation of Markov Games Through Information Design","Tao Zhang and Quanyan Zhu","2021","paper","arXiv preprint","arxiv.org/abs/2102.07152",0,"","agents"],["Interactive Learning from Activity Description","Khanh Nguyen and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.07024",0,"","evals agents"],["Mitigating Negative Side Effects via Environment Shaping","Sandhya Saisubramanian and Shlomo Zilberstein","2021","paper","arXiv preprint","arxiv.org/abs/2102.07017",0,"","rlhf evals agents"],["Modelling Cooperation in Network Games with Spatio-Temporal Complexity","Michiel A. Bakker and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.06911",0,"","deception agents"],["Weak identifiability and its consequences in strategic settings","Jesse Clifton","2021","report","longtermrisk.org","longtermrisk.org/weak-identifiability-and-its-consequences-in-strategic-settings/",0,"","agents"],["A Decentralized Approach towards Responsible AI in Social Ecosystems","Wenjing Chu","2021","paper","arXiv preprint","arxiv.org/abs/2102.06362",0,"","governance robustness"],["Discovery of Options via Meta-Learned Subgoals","Vivek Veeriah and 8 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.06741",0,"","agents policy"],["Explaining Neural Scaling Laws","Yasaman Bahri and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.06701",0,"","training-data scaling-laws"],["Mapping the Conceptual Territory in AI Existential Safety and Alignment","jbkjr","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HEZgGBZTpT4Bov7nH/mapping-the-conceptual-territory-in-ai-existential-safety",0,"","scalable-oversight"],["Tournesol, YouTube and AI Risk","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/8q2ySr7yxx7MSR35i/tournesol-youtube-and-ai-risk",0,"",""],["Institute for Assured Autonomy (IAA) newsletter","Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GtEpGu93zsLuZSSZS/institute-for-assured-autonomy-iaa-newsletter",0,"",""],["Self-Organising Textures","Eyvind Niklasson and 2 others","2021","report","Distill","distill.pub/selforg/2021/textures",0,"",""],["Stuart Russell Human Compatible AI Roundtable with Allan Dafoe, Rob Reich, & Marietje Schaake","Mahendra Prasad","2021","blog","EA Forum","forum.effectivealtruism.org/posts/2ENbqRr9Q7PSABtv2/stuart-russell-human-compatible-ai-roundtable-with-allan",0,"","governance policy"],["[AN #137]: Quantifying the benefits of pretraining on downstream task performance","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/85HgXZvNdTdfRJhar/an-137-quantifying-the-benefits-of-pretraining-on-downstream",0,"",""],["Language models are 0-shot interpreters","janus","2021","blog","generative.ink","generative.ink/posts/language-models-are-0-shot-interpreters/",0,"",""],["Some global catastrophic risk estimates","Tamay","2021","blog","EA Forum","forum.effectivealtruism.org/posts/27aXsJRRAoNZFw9K3/some-global-catastrophic-risk-estimates",0,"","forecasting"],["Transfer Reinforcement Learning across Homotopy Classes","Zhangjie Cao and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.05207",0,"","agents"],["Epistemology of HCH","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CDSXoC54CjbXQNLGr/epistemology-of-hch",0,"",""],["Equilibrium Refinements for Multi-Agent Influence Diagrams: Theory and Practice","Lewis Hammond and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.05008",0,"","agents"],["Fixing The Good Regulator Theorem","johnswentworth","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Dx9LoqsEh3gHNJMDk/fixing-the-good-regulator-theorem",0,"","robustness training-data"],["Loom: interface to the multiverse","janus","2021","blog","generative.ink","generative.ink/posts/loom-interface-to-the-multiverse/",0,"",""],["13 Recent Publications on Existential Risk (Jan 2021 update)","HaydnBelfield","2021","blog","EA Forum","forum.effectivealtruism.org/posts/3ykdme7ka2NBo32Qe/13-recent-publications-on-existential-risk-jan-2021-update",0,"","governance"],["Alchemical marriage: GPT-3 x CLIP","janus","2021","blog","generative.ink","generative.ink/posts/alchemical-marriage-gpt-3-x-clip/",0,"",""],["Learning Curve Theory","Marcus Hutter","2021","paper","Latest 2021 version at http://www.hutter1.net/publ/scaling.pdf","arxiv.org/abs/2102.04074",0,"","scaling-laws"],["Playing the Blame Game with Robots","Markus Kneer and Michael T. Stuart","2021","paper","arXiv preprint","arxiv.org/abs/2102.04527",0,"",""],["This Museum Does Not Exist: GPT-3 x CLIP","janus","2021","blog","generative.ink","generative.ink/posts/this-museum-does-not-exist-gpt-3-x-clip/",0,"",""],["Consequences of Misaligned AI","Simon Zhuang and Dylan Hadfield-Menell","2021","paper","NeurIPS 2020","arxiv.org/abs/2102.03896",0,"","agents"],["Timeline of AI safety","riceissa","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/SEfjw57Qw8mCzy36n/timeline-of-ai-safety",0,"","forecasting"],["Reflections on Artificial Intelligence for Humanity","Bertrand Braunschweig and Malik Ghallab","2021","report","goodreads.com","www.goodreads.com/book/show/56634187-reflections-on-artificial-intelligence-for-humanity",0,"",""],["AI Can Stop Mass Shootings, and More","Selmer Bringsjord and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.09343",0,"","agents"],["Creating AGI Safety Interlocks","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/BZKLf629NDNfEkZzJ/creating-agi-safety-interlocks",0,"",""],["Evolutions Building Evolutions: Layers of Generate and Test","plex","2021","blog","LessWrong","www.lesswrong.com/posts/HDAjZaeTtEYyDk93a/evolutions-building-evolutions-layers-of-generate-and-test",0,"",""],["Learning Normativity: Language","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/5eY6A4Zfu6rfeMfJS/learning-normativity-language",0,"",""],["AI Development for the Public Interest: From Abstraction Traps to Sociotechnical Risks","McKane Andrus and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.04255",0,"",""],["Exploring Beyond-Demonstrator via Meta Learning-Based Reward Extrapolation","Mingqi Yuan and Mao-on Pun","2021","paper","arXiv preprint","arxiv.org/abs/2102.02454",0,"","policy training-data"],["Feedback in Imitation Learning: The Three Regimes of Covariate Shift","Jonathan Spencer and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.02872",0,"","benchmarks"],["OpenAI: \"Scaling Laws for Transfer\", Hernandez et al.","Lukas Finnveden","2021","blog","LessWrong","www.lesswrong.com/posts/g3DvR7iFN7jE7nKEL/openai-scaling-laws-for-transfer-hernandez-et-al",0,"","scaling-laws"],["Understanding the Capabilities, Limitations, and Societal Impact of Large Language Models","Alex Tamkin and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.02503",0,"","policy"],["Visualizing Weights","Chelsea Voss and 6 others","2021","report","Distill","distill.pub/2020/circuits/visualizing-weights",0,"",""],["[AN #136]: How well will GPT-N perform on downstream tasks?","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HJMQg8MksHq5ipDpN/an-136-how-well-will-gpt-n-perform-on-downstream-tasks",0,"",""],["A Critique of Non-Obstruction","Joe_Collman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZqfT5xTuNf6okrepY/a-critique-of-non-obstruction",0,"",""],["Counterfactual Planning in AGI Systems","Koen.Holtman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7EnZgaepSBwaZXA5y/counterfactual-planning-in-agi-systems",0,"","theory"],["Distinguishing claims about training vs deployment","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/L9HcyaiWBLYe7vXid/distinguishing-claims-about-training-vs-deployment",0,"","goodharts-law instrumental-convergence agents"],["GPT-3 x CLIP worldbuilding","janus","2021","blog","generative.ink","generative.ink/posts/gpt-3-x-clip-worldbuilding/",0,"",""],["Agent Incentives: A Causal Perspective","Tom Everitt and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.01685",0,"","evals agents"],["Data, Architecture, or Losses: What Contributes Most to Multimodal Transformer Success?","Aida Nematzadeh and 4 others","2021","blog","deepmind.com","www.deepmind.com/blog/data-architecture-or-losses-what-contributes-most-to-multimodal-transformer-success",0,"",""],["Scaling Laws for Transfer","Danny Hernandez and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2102.01293",0,"","training-data scaling-laws"],["AI Verification: Mechanisms to Ensure AI Arms Control Compliance","Matthew Mittelsteadt","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ai-verification/",0,"",""],["CHAI Newsletter #3 2020","CHAI","2021","report","drive.google.com","drive.google.com/file/d/1MsH109azMhvGFh9OFTlwPm478naO1L2v/view?usp=sharing",0,"",""],["CLIP hallucinates 1900-2030","janus","2021","blog","generative.ink","generative.ink/posts/clip-hallucinates-1900-2030/",0,"",""],["Institutionalizing ethics in AI through broader impact requirements","Carina E. A. Prunkl and 5 others","2021","report","nature.com","www.nature.com/articles/s42256-021-00298-y",0,"",""],["Trusted Partners","Margarita Konaev and 2 others","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/trusted-partners/",0,"",""],["CLIP art","janus","2021","blog","generative.ink","generative.ink/posts/clip-art/",0,"",""],["Fairness through Social Welfare Optimization","Violet Xinying Chen and J. N. Hooker","2021","paper","arXiv preprint","arxiv.org/abs/2102.00311",0,"",""],["Limiting Causality by Complexity Class","Bunthut","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QFuypcQGZK59TaKos/limiting-causality-by-complexity-class",0,"",""],["AMA on EA Forum: Ajeya Cotra, researcher at Open Phil","Ajeya Cotra","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/F2C6KKRXGeZ424mi7/ama-on-ea-forum-ajeya-cotra-researcher-at-open-phil",0,"",""],["Challenges for Using Impact Regularizers to Avoid Negative Side Effects","David Lindner and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.12509",0,"","agents"],["Counterfactual Planning in AGI Systems","Koen Holtman","2021","paper","arXiv preprint","arxiv.org/abs/2102.00834",0,"","agents"],["AMA: Ajeya Cotra, researcher at Open Phil","Ajeya","2021","blog","EA Forum","forum.effectivealtruism.org/posts/QAqghTmp7FSMcJ4ch/ama-ajeya-cotra-researcher-at-open-phil",0,"","forecasting"],["Extracting Money from Causal Decision Theorists","Caspar Oesterheld","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/xPeWJaAzp2LeDdP4Z/extracting-money-from-causal-decision-theorists",0,"",""],["Making Responsible AI the Norm rather than the Exception","Abhishek Gupta","2021","paper","arXiv preprint","arxiv.org/abs/2101.11832",0,"",""],["[AN #135]: Five properties of goal-directed systems","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wJYitLpqujQqwX7ke/an-135-five-properties-of-goal-directed-systems",0,"",""],["High-Low Frequency Detectors","Ludwig Schubert and 4 others","2021","report","Distill","distill.pub/2020/circuits/frequency-edges",0,"",""],["January 2021 Newsletter","Rob Bensinger","2021","blog","intelligence.org","intelligence.org/2021/01/27/january-2021-newsletter/",0,"",""],["Optimal play in human-judged Debate usually won't answer your question","Joe_Collman","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/35748mXjzwxDrX7yQ/optimal-play-in-human-judged-debate-usually-won-t-answer",0,"",""],["Muppet: Massive Multi-task Representations with Pre-Finetuning","Armen Aghajanyan and 5 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.11038",0,"",""],["Accumulating Risk Capital Through Investing in Cooperation","Charlotte Roman and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.10305",0,"","evals agents"],["Language models are multiverse generators","janus","2021","blog","generative.ink","generative.ink/posts/language-models-are-multiverse-generators/",0,"",""],["Measuring Intelligence and Growth Rate: Variations on Hibbard's Intelligence Measure","Samuel Alexander and Bill Hibbard","2021","paper","Journal of Artificial General Intelligence 12(1), 2021","arxiv.org/abs/2101.12047",0,"","agents"],["What is a VNM stable set, really?","Nisan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/CLuCgA2Ab7sBfvEuW/what-is-a-vnm-stable-set-really",0,"",""],["FC final: Can Factored Cognition schemes scale?","Rafael Harth","2021","blog","LessWrong","www.lesswrong.com/posts/qXgge5EYGRvddSMke/fc-final-can-factored-cognition-schemes-scale",0,"",""],["The Internet, mirrored by GPT-3","janus","2021","blog","generative.ink","generative.ink/posts/the-internet-mirrored-by-gpt-3/",0,"",""],["[Podcast] Ajeya Cotra on worldview diversification and how big the future could be","BrownHairedEevee","2021","blog","EA Forum","forum.effectivealtruism.org/posts/CnD4fHwkgnknbz3ED/podcast-ajeya-cotra-on-worldview-diversification-and-how-big",0,"","governance"],["Baobao Zhang: How social science research can inform AI governance","EA Global","2021","blog","EA Forum","forum.effectivealtruism.org/posts/9kNqYzEAYtvLg2BbR/baobao-zhang-how-social-science-research-can-inform-ai",0,"","governance"],["Communicating Clearly","Tamsin Leake","2021","blog","carado.moe","carado.moe/communicating-clearly.html",0,"",""],["Poll: Which variables are most strategically relevant?","Daniel Kokotajlo and Noa Nabeshima","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/yhb5BNksWcESezp7p/poll-which-variables-are-most-strategically-relevant",0,"","forecasting"],["[AN #134]: Underspecification as a cause of fragility to distribution shift","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/nM99oLhRzrmLWozoM/an-134-underspecification-as-a-cause-of-fragility-to",0,"","robustness"],["Counterfactual control incentives","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/67a8C6KsKn2NyW2Ry/counterfactual-control-incentives",0,"",""],["Singapore AI Policy Career Guide","Yi-Yang","2021","blog","EA Forum","forum.effectivealtruism.org/posts/umeMcbD4jDseLjsgT/singapore-ai-policy-career-guide",0,"","governance policy"],["Infra-Bayesianism Unwrapped","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zi7nmuSmBFbQWgFBa/infra-bayesianism-unwrapped",0,"","theory"],["Shielding Atari Games with Bounded Prescience","Mirco Giacobbe and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.08153",0,"","evals agents"],["UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","Siyi Hu and 3 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.08001",0,"","agents policy"],["Against the Backward Approach to Goal-Directedness","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/adKSWktLbxfihDANM/against-the-backward-approach-to-goal-directedness",0,"",""],["Some thoughts on risks from narrow, non-agentic AI","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/AWbtbmC6rAg6dh75b/some-thoughts-on-risks-from-narrow-non-agentic-ai",0,"","agents"],["Some thoughts on risks from narrow, non-agentic AI","richard_ngo","2021","blog","EA Forum","forum.effectivealtruism.org/posts/RP2JXebirXqeaQqH6/some-thoughts-on-risks-from-narrow-non-agentic-ai",0,"","agents governance"],["Birds, Brains, Planes, and AI: Against Appeals to the Complexity/Mysteriousness/Efficiency of the Brain","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/HhWhaSzQr6xmBki8F/birds-brains-planes-and-ai-against-appeals-to-the-complexity",0,"","forecasting"],["Birds, Brains, Planes, and AI: Against Appeals to the Complexity/Mysteriousness/Efficiency of the Brain","kokotajlod","2021","blog","EA Forum","forum.effectivealtruism.org/posts/m6zJ8xTuQp398uopy/birds-brains-planes-and-ai-against-appeals-to-the-complexity",0,"","governance forecasting"],["Literature Review on Goal-Directedness","adamShimi and 2 others","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/cfXwr6NC9AqZ9kr8g/literature-review-on-goal-directedness",0,"","agents"],["Short summary of mAIry's room","Stuart_Armstrong","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/rmBS5nTJh6pxERWEu/short-summary-of-mairy-s-room",0,"",""],["Adversarial Interaction Attack: Fooling AI to Misinterpret Human Intentions","Nodens Koren and 4 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.06704",0,"","agents"],["Excerpt from Arbital Solomonoff induction dialogue","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/wsBpJn7HWEPCJxYai/excerpt-from-arbital-solomonoff-induction-dialogue",0,"",""],["Teaming up with information agents","Jurriaan van Diggelen and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.06133",0,"","agents"],["The Challenge of Value Alignment: from Fairer Algorithms to AI Safety","Iason Gabriel and Vafa Ghazavi","2021","paper","arXiv preprint","arxiv.org/abs/2101.06060",0,"",""],["Why I'm excited about Debate","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/LDsSqXf9Dpu3J3gHD/why-i-m-excited-about-debate",0,"",""],["[link] Centre for the Governance of AI 2020 Annual Report","MarkusAnderljung","2021","blog","EA Forum","forum.effectivealtruism.org/posts/hfZiAMKMLYw6Yoms5/link-centre-for-the-governance-of-ai-2020-annual-report",0,"","governance"],["A canonical bit-encoding for ranged integers","Tamsin Leake","2021","blog","carado.moe","carado.moe/canonical-bit-varints.html",0,"",""],["Evaluating the Robustness of Collaborative Agents","Paul Knott and 6 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.05507",0,"","evals agents robustness"],["Thoughts on Iason Gabriel’s Artificial Intelligence, Values, and Alignment","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Z2rkdEAJ9MvYPBeYW/thoughts-on-iason-gabriel-s-artificial-intelligence-values",0,"",""],["[AN #133]: Building machines that can cooperate (with humans, institutions, or other machines)","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/S8khsrXnHEwYbhd8X/an-133-building-machines-that-can-cooperate-with-humans",0,"",""],["Some recent survey papers on (mostly near-term) AI safety, security, and assurance","Aryeh Englander","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/GTcWrenvDMsThTQ26/some-recent-survey-papers-on-mostly-near-term-ai-safety",0,"","assurance"],["AGI Safety and Alignment with Robert Miles-by Machine Ethics-date 20210113","Robert Miles","2021","report","drive.google.com","drive.google.com/file/d/1Sa_PTmksYLvEAwoPsspbGNn6ZFVMeXnC/view?usp=share_link",0,"",""],["AI and International Stability: Risks and Confidence-Building Measures","Michael Horowitz and Paul Scharre","2021","report","cnas.org","www.cnas.org/publications/reports/ai-and-international-stability-risks-and-confidence-building-measures",0,"","interpretability evals agents"],["How should we invest in \"long-term short-termism\" given the likelihood of transformative AI?","James_Banks","2021","blog","EA Forum","forum.effectivealtruism.org/posts/s29BdN8EeyKjg5v6M/how-should-we-invest-in-long-term-short-termism-given-the",0,"","robustness"],["Methods of prompt programming","janus","2021","blog","generative.ink","generative.ink/posts/methods-of-prompt-programming/",0,"",""],["Review of 'Debate on Instrumental Convergence between LeCun, Russell, Bengio, Zador, and More'","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZPEGLoWMN242Dob6g/review-of-debate-on-instrumental-convergence-between-lecun",0,"","instrumental-convergence"],["The Immigration Preferences of Top AI Researchers: New Survey Evidence | GovAI","Remco Zwetsloot and 4 others","2021","report","governance.ai","www.governance.ai/research-paper/the-immigration-preferences-of-top-ai-researchers-new-survey-evidence",0,"",""],["Transparency and AGI safety","jylin04","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/QirLfXhDPYWCP8PK5/transparency-and-agi-safety",0,"","interpretability mechanistic-interpretability forecasting robustness"],["Imitative Generalisation (AKA 'Learning the Prior')","Beth Barnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/JKj5Krff5oKMb8TjT/imitative-generalisation-aka-learning-the-prior-1",0,"","scalable-oversight interpretability robustness training-data"],["Prediction can be Outer Aligned at Optimum","Lukas Finnveden","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/3D2MGF2fZhWSNb7aw/prediction-can-be-outer-aligned-at-optimum",0,"",""],["Review of Soft Takeoff Can Still Lead to DSA","Daniel Kokotajlo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/P448hmmAeGepQDREs/review-of-soft-takeoff-can-still-lead-to-dsa",0,"","forecasting"],["The Case for a Journal of AI Alignment","adamShimi","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/hNNM6gP5yZcHffmpD/the-case-for-a-journal-of-ai-alignment",0,"",""],["What does it mean to become an expert in AI Hardware?","Toph","2021","blog","EA Forum","forum.effectivealtruism.org/posts/HrS2pXQ3zuTwr2SKS/what-does-it-mean-to-become-an-expert-in-ai-hardware-1",0,"","governance compute-governance"],["Bridging In- and Out-of-distribution Samples for Their Better Discriminability","Engkarat Techapanurak and 2 others","2021","paper","arXiv preprint","arxiv.org/abs/2101.02500",0,"","benchmarks deception robustness"],["Eight claims about multi-agent AGI safety","Richard_Ngo","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/dSAJdi99XmqftqXXq/eight-claims-about-multi-agent-agi-safety",0,"","agents"],["Solving for X?' Towards a Problem-Finding Framework to Ground Long-Term Governance Strategies for Artificial Intelligence","Hin-Yan Liu and Matthijs M. Maas","2021","report","papers.ssrn.com","papers.ssrn.com/abstract=3761623",0,"","governance"],["[AN #132]: Complex and subtly incorrect arguments as an obstacle to debate","Rohin Shah","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/huNvfttDpxCApC3xZ/an-132-complex-and-subtly-incorrect-arguments-as-an-obstacle",0,"",""],["Legal Priorities Research: A Research Agenda","jonasschuett and Legal Priorities Project","2021","blog","EA Forum","forum.effectivealtruism.org/posts/XpwejKTZkRbJ5s4cp/legal-priorities-research-a-research-agenda",0,"","governance policy"],["Review of 'But exactly how complex and fragile?'","TurnTrout","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/r6p5cqT6aWYGCYHJx/review-of-but-exactly-how-complex-and-fragile",0,"",""],["The National Defense Authorization Act Contains AI Provisions","ryan_b","2021","blog","LessWrong","www.lesswrong.com/posts/6Nuw7mLc6DjRY4mwa/the-national-defense-authorization-act-contains-ai",0,"","governance"],["The Pointers Problem: Clarifications/Variations","abramdemski","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/7Zn4BwgsiPFhdB6h8/the-pointers-problem-clarifications-variations",0,"",""],["Existential risks from a Thomist Christian perspective","Stefan Riedener","2021","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/stefan-riedener-existential-risks-from-a-thomist-christian-perspective/",0,"",""],["Multi-dimensional rewards for AGI interpretability and control","Steven Byrnes","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/Lj9QXcqkcuR4iHJ7Q/multi-dimensional-rewards-for-agi-interpretability-and",0,"","interpretability"],["2020-21 New Year review","Victoria Krakovna","2021","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2021/01/03/2020-21-new-year-review/",0,"",""],["Are we all misaligned?","Mateusz Mazurkiewicz","2021","blog","LessWrong","www.lesswrong.com/posts/Lshuoww97Loy2h7kw/are-we-all-misaligned-1",0,"",""],["Mental subagent implications for AI Safety","moridinamael","2021","blog","LessWrong","www.lesswrong.com/posts/datp9aq4DAzEP8taM/mental-subagent-implications-for-ai-safety",0,"","agents"],["A General Counterexample to Any Decision Theory and Some Responses","Joar Skalse","2021","paper","arXiv preprint","arxiv.org/abs/2101.00280",0,"","theory"],["AI & Antitrust: Reconciling Tensions Between Competition Law and Cooperative AI Development | Yale Journal of Law & Technology","Shin-Shin Hua and Haydn Belfield","2021","report","yjolt.org","yjolt.org/ai-antitrust-reconciling-tensions-between-competition-law-and-cooperative-ai-development",0,"",""],["AI Alignment, Philosophical Pluralism, and the Relevance of Non-Western Philosophy","xuan","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/jS2iiDPqMvZ2tnik2/ai-alignment-philosophical-pluralism-and-the-relevance-of",0,"",""],["AI and the Future of Cyber Competition","Wyatt Hoffman","2021","report","cset.georgetown.edu","cset.georgetown.edu/publication/ai-and-the-future-of-cyber-competition/",0,"",""],["AI CERTIFICATION: Advancing Ethical Practice by Reducing Information Asymmetries","Peter Cihon and 3 others","2021","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/9427056/",0,"","assurance"],["Artificial Canaries: Early Warning Signs for Anticipatory and Democratic Governance of AI","Carla Zoe Cremer and Jess Whittlestone","2021","report","ijimai.org","www.ijimai.org/journal/sites/default/files/2021-02/ijimai_6_5_10.pdf",0,"","governance"],["Artificial Intelligence Governance Under Change: Foundations, Facets, Frameworks","Matthijs M. Maas","2021","report","researchgate.net","www.researchgate.net/publication/351314988_Artificial_Intelligence_Governance_Under_Change_Foundations_Facets_Frameworks",0,"","governance"],["Emerging Technologies: More to explore","EA Handbook","2021","blog","EA Forum","forum.effectivealtruism.org/posts/yasigF54XKCzuxcfh/emerging-technologies-more-to-explore",0,"",""],["Negative Side Effects and AI Agent Indicators: Experiments in SafeLife","John Burden and Jose Hernandez-Orallo","2021","report","josephorallo.webs.upv.es","josephorallo.webs.upv.es/escrits/SafeAI2021.pdf",0,"","agents"],["Non-solving ideologies","Tamsin Leake","2021","blog","carado.moe","carado.moe/nonsolving-ideologies.html",0,"",""],["Not a paper, but I find Chris Olah’s interview on the 80,000 Hours podcast super inspiring","Rob Wiblin and Chris Olah","2021","report","80000hours.org","80000hours.org/podcast/episodes/chris-olah-interpretability-research/",0,"","interpretability mechanistic-interpretability evals robustness"],["QNRs: Toward Language for Intelligent Machines","K. Eric Drexler","2021","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/2021/08/QNRs_FHI-TR-2021-3.0.pdf",0,"",""],["Reflections on Larks’ 2020 AI alignment literature review","Alex Flint","2021","blog","AI Alignment Forum","www.alignmentforum.org/posts/uEo4Xhp7ziTKhR6jq/reflections-on-larks-2020-ai-alignment-literature-review",0,"",""],["Safe Pareto Improvements for Delegated Game Playing","Caspar Oesterheld and Vincent Conitzer","2021","report","users.cs.duke.edu","users.cs.duke.edu/~conitzer/safeAAMAS21.pdf",0,"",""],["Safe Pareto Improvements for Delegated Game Playing","Caspar Oesterheld and Vincent Conitzer","2021","report","cs.cmu.edu","www.cs.cmu.edu/~15784/SPI.pdf",0,"",""],["Socially Responsible AI Algorithms: Issues, Purposes, and Challenges","Lu Cheng and 2 others","2021","paper","Journal of Artificial Intelligence Research 71 (2021) 1137-1181","arxiv.org/abs/2101.02032",0,"",""],["The case against economic values in the orbitofrontal cortex (or anywhere else in the brain)","Benjamin Hayden and Yael Niv","2021","report","psyarxiv.com","psyarxiv.com/7hgup/",0,"","policy"],["The Ethics of Sustainability for Artificial Intelligence","Andrea Owe and Seth Baum","2021","report","gcrinstitute.org","gcrinstitute.org/the-ethics-of-sustainability-for-artificial-intelligence/",0,"",""],["Truthful AI: Developing and governing AI that does not lie","Owain Evans1† and 7 others","2021","paper","arXiv preprint","arxiv.org/abs/2110.06674",0,"",""],["TruthfulQA: Measuring How Models Mimic Human Falsehoods","Stephanie Lin","2021","paper","arXiv preprint","arxiv.org/abs/2109.07958",0,"","evals benchmarks robustness training-data"],["WHAT IS THE UPPER LIMIT OF VALUE?","Anders Sandberg and David Manheim","2021","report","philpapers.org","philpapers.org/archive/MANWIT-6.pdf",0,"",""],["2020 Survey of Artificial General Intelligence Projects for Ethics, Risk, and Policy | Global Catastrophic Risk Institute","Seth Baum","2020","report","gcrinstitute.org","gcrinstitute.org/2020-survey-of-artificial-general-intelligence-projects-for-ethics-risk-and-policy/",0,"","policy"],["[AN #131]: Formalizing the argument of ignored attributes in a utility function","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/k2sBrR4gJX9BNTuoa/an-131-formalizing-the-argument-of-ignored-attributes-in-a",0,"",""],["Core values: Defining freedom","Tamsin Leake","2020","blog","carado.moe","carado.moe/defining-freedom.html",0,"",""],["December 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/12/30/december-2020-newsletter/",0,"",""],["A canonical and efficient byte-encoding for ints","Tamsin Leake","2020","blog","carado.moe","carado.moe/canonical-byte-varints.html",0,"",""],["Against GDP as a metric for timelines and takeoff speeds","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/aFaKhG86tTrKvtAnT/against-gdp-as-a-metric-for-timelines-and-takeoff-speeds",0,"","forecasting"],["Against GDP as a metric for timelines and takeoff speeds","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/NgBQcZbMtDLW8fpSg/against-gdp-as-a-metric-for-timelines-and-takeoff-speeds",0,"","governance forecasting"],["AXRP Episode 1 - Adversarial Policies with Adam Gleave","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8MZ72PYa3kRe4yRDD/axrp-episode-1-adversarial-policies-with-adam-gleave",0,"","robustness"],["AXRP Episode 2 - Learning Human Biases with Rohin Shah","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BJAcnMBHGua3tFKu5/axrp-episode-2-learning-human-biases-with-rohin-shah",0,"",""],["AXRP Episode 3 - Negotiable Reinforcement Learning with Andrew Critch","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/u7o7HtChnZ5x8SqvA/axrp-episode-3-negotiable-reinforcement-learning-with-andrew",0,"",""],["Dario Amodei leaves OpenAI","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/7r8KjgqeHaYDzJvzF/dario-amodei-leaves-openai",0,"","governance"],["Debate Minus Factored Cognition","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/a2NZr87sGYpXhzsth/debate-minus-factored-cognition",0,"",""],["Multi-Principal Assistance Games: Definition and Collegial Mechanisms","Arnaud Fickinger and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.14536",0,"","agents"],["Why Neural Networks Generalise, and Why They Are (Kind of) Bayesian","Joar Skalse","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YSFJosoHYFyXjoYWa/why-neural-networks-generalise-and-why-they-are-kind-of",0,"",""],["You are your information system","Tamsin Leake","2020","blog","carado.moe","carado.moe/you-are-your-information-system.html",0,"",""],["[AN #130]: A new AI x-risk podcast, and reviews of the field","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/h2ipMwfx4D3oenzu2/an-130-a-new-ai-x-risk-podcast-and-reviews-of-the-field",0,"",""],["CSS for pixeley images","Tamsin Leake","2020","blog","carado.moe","carado.moe/css-for-pixeley-images.html",0,"",""],["Defusing AGI Danger","Mark Xu","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BSrfDWpHgFpzGRwJS/defusing-agi-danger",0,"",""],["Operationalizing compatibility with strategy-stealing","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WwJdaymwKq6qyJqBX/operationalizing-compatibility-with-strategy-stealing",0,"",""],["2019 Review Rewrite: Seeking Power is Often Robustly Instrumental in MDPs","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/mxXcPzpgGx4f8eK7v/2019-review-rewrite-seeking-power-is-often-robustly",0,"","instrumental-convergence"],["Announcing AXRP, the AI X-risk Research Podcast","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/NWi8ztKCbguBEAwdG/announcing-axrp-the-ai-x-risk-research-podcast-1",0,"",""],["Antitrust and Artificial Intelligence (AAI): Antitrust Vigilance Lifecycle and AI Legal Reasoning Autonomy","Lance Eliot","2020","paper","arXiv preprint","arxiv.org/abs/2012.13016",0,"","monitoring"],["Augmenting Policy Learning with Routines Discovered from a Single Demonstration","Zelin Zhao and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.12469",0,"","benchmarks policy"],["Debate update: Obfuscated arguments problem","Beth Barnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PJLABqQ962hZEqhdB/debate-update-obfuscated-arguments-problem",0,"","scalable-oversight debate deception"],["Unfair feedback loops","Tamsin Leake","2020","blog","carado.moe","carado.moe/unfair-feedback-loops.html",0,"",""],["AI Impacts 2020 review","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/ai-impacts-2020-review/",0,"",""],["CFP for the Largest Annual Meeting of Political Science: Get Help With Your Research Submission","Mahendra Prasad","2020","blog","EA Forum","forum.effectivealtruism.org/posts/hfCLvey5JsxkD6k9n/cfp-for-the-largest-annual-meeting-of-political-science-get",0,"",""],["Rationalist by necessity","Tamsin Leake","2020","blog","carado.moe","carado.moe/rationalist-by-necessity.html",0,"",""],["TAI Safety Bibliographic Database","JessRiedel","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4DegbDJJiMX2b3EKm/tai-safety-bibliographic-database",0,"",""],["TAI Safety Bibliographic Database","Jess_Riedel","2020","blog","EA Forum","forum.effectivealtruism.org/posts/S7x3ztfd9h8ux68wN/tai-safety-bibliographic-database",0,"","governance"],["2020 AI Alignment Literature Review and Charity Comparison","Larks","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pTYDdcag9pTzFQ7vw/2020-ai-alignment-literature-review-and-charity-comparison",0,"",""],["2020 Updates and Strategy","Malo Bourgon","2020","blog","intelligence.org","intelligence.org/2020/12/21/2020-updates-and-strategy/",0,"",""],["Against Unicode","Tamsin Leake","2020","blog","carado.moe","carado.moe/against-unicode.html",0,"",""],["Evaluating Agents without Rewards","Brendon Matusch and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.11538",0,"","evals agents"],["How Lesswrong helped me make $25K: A rational pricing strategy","kareemabukhadra","2020","blog","LessWrong","www.lesswrong.com/posts/aKT6WCs3ASvBWfLw9/how-lesswrong-helped-me-make-usd25k-a-rational-pricing",0,"","theory"],["Taking Principles Seriously: A Hybrid Approach to Value Alignment","Tae Wan Kim and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.11705",0,"",""],["Cringe as prejudice","Tamsin Leake","2020","blog","carado.moe","carado.moe/cringe-as-prejudice.html",0,"",""],["Hierarchical planning: context agents","Charlie Steiner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6ayQbR5opoTN4AgFb/hierarchical-planning-context-agents",0,"","agents"],["Probabilistic Dependency Graphs","Oliver Richardson and Joseph Y Halpern","2020","paper","arXiv preprint","arxiv.org/abs/2012.10800",0,"",""],["Exploring Fluent Query Reformulations with Text-to-Text Transformers and Reinforcement Learning","Jerry Zikun Chen and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.10033",0,"","evals policy"],["Extrapolating GPT-N performance","Lukas Finnveden","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/k2SNji3jXaLGhBeYP/extrapolating-gpt-n-performance",0,"","forecasting"],["AI Impacts key questions of interest","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/ai-impacts-key-questions-of-interest/",0,"",""],["Can Transformers Reason About Effects of Actions?","Pratyay Banerjee and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.09938",0,"","evals deception"],["Extracting and Using Preference Information from the State of the World","Rohin Monish Shah","2020","report","www2.eecs.berkeley.edu","www2.eecs.berkeley.edu/Pubs/TechRpts/2020/EECS-2020-210.pdf",0,"",""],["How long till Inverse AlphaFold?","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/QqmPkJzDwPytejt4w/how-long-till-inverse-alphafold",0,"","forecasting"],["Mapping the Conceptual Territory in AI Existential Safety and Alignment","Jack Koch","2020","report","jbkjr.me","jbkjr.me/posts/2020/12/mapping_conceptual_territory_AI_safety_alignment/",0,"",""],["Mitigating x-risk through modularity","Toby Newberry","2020","blog","EA Forum","forum.effectivealtruism.org/posts/nTZ6bnm8HFjjJWBmt/mitigating-x-risk-through-modularity",0,"",""],["Open Philanthropy's AI governance grantmaking (so far)","Aaron Gertler","2020","blog","EA Forum","forum.effectivealtruism.org/posts/kZqvjtLMkQyByi6yb/open-philanthropy-s-ai-governance-grantmaking-so-far",0,"","governance forecasting"],["[AN #129]: Explaining double descent by measuring bias and variance","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/r3AcHkAXPbjPwXFjc/an-129-explaining-double-descent-by-measuring-bias-and",0,"",""],["Homogeneity vs. heterogeneity in AI takeoff scenarios","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/mKBfa8v4S9pNKSyKK/homogeneity-vs-heterogeneity-in-ai-takeoff-scenarios",0,"","forecasting"],["Less Basic Inframeasure Theory","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/idP5E5XhJGh9T5Yq9/less-basic-inframeasure-theory",0,"",""],["Our AI governance grantmaking so far","Luke Muehlhauser","2020","report","openphilanthropy.org","www.openphilanthropy.org/blog/ai-governance-grantmaking",0,"","assurance governance robustness"],["Challenges of Aligning Artificial Intelligence with Human Values","Margit Sutrop","2020","report","ies.ee","www.ies.ee/bahps/acta-baltica/abhps-8-2/04_Sutrop-2020-2-04.pdf",0,"",""],["Draft report on AI timelines","Ajeya","2020","blog","EA Forum","forum.effectivealtruism.org/posts/ZkuiHKjPWsjf5zTrw/draft-report-on-ai-timelines",0,"","forecasting"],["Risk Map of AI Systems","VojtaKovarik and Jan_Kulveit","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QskBy5uDd2oeEGkBB/risk-map-of-ai-systems",0,"",""],["Some AI research areas and their relevance to existential safety","Andrew Critch","2020","blog","EA Forum","forum.effectivealtruism.org/posts/j8XuuBvFhsKdivv8Q/some-ai-research-areas-and-their-relevance-to-existential",0,"","governance"],["Efficient Querying for Cooperative Probabilistic Commitments","Qi Zhang and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.07195",0,"","agents"],["Extracting Training Data from Large Language Models","Nicholas Carlini and 11 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.07805",0,"","evals training-data"],["What are the best precedents for industries failing to invest in valuable AI research?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/27h99G7P6fkucKdkk/what-are-the-best-precedents-for-industries-failing-to",0,"",""],["Wilds: A Benchmark of in-the-Wild Distribution Shifts","Pang Wei Koh and Shiori Sagawa","2020","paper","arXiv preprint","arxiv.org/abs/2012.07421",0,"","evals benchmarks robustness monitoring"],["Clarifying Factored Cognition","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eCWkJrFff7oMLwjEp/clarifying-factored-cognition",0,"",""],["Avoiding Side Effects in Complex Environments","TurnTrout and nealeratzlaff","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/5kurn5W62C5CpSWq6/avoiding-side-effects-in-complex-environments",0,"","benchmarks agents"],["Imitating Interactive Intelligence","Josh Abramson and 23 others","2020","blog","deepmind.com","www.deepmind.com/blog/imitating-interactive-intelligence",0,"",""],["Interdisciplinary Approaches to Understanding Artificial Intelligence's Impact on Society","Suresh Venkatasubramanian and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.06057",0,"","monitoring"],["How energy efficient are human-engineered flight designs relative to natural ones?","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/are-human-engineered-flight-designs-better-or-worse-than-natural-ones/",0,"","forecasting"],["Imitating Interactive Intelligence","Josh Abramson and 28 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.05672",0,"","evals agents policy"],["Learning to Resolve Conflicts for Multi-Agent Path Finding with Conflict-Based Search","Taoan Huang and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.06005",0,"","benchmarks agents"],["Neurosymbolic AI: The 3rd Wave","Artur d'Avila Garcez and Luis C. Lamb","2020","paper","arXiv preprint","arxiv.org/abs/2012.05876",0,"","interpretability evals"],["What technologies could cause world GDP doubling times to be <8 years?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2rQ9vv9HY6i2Z2vQ4/what-technologies-could-cause-world-gdp-doubling-times-to-be",0,"",""],["[AN #128]: Prioritizing research on AI existential safety based on its application to governance demands","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/esjMWREvj3WKZpBZd/an-128-prioritizing-research-on-ai-existential-safety-based",0,"","governance"],["What the AI Community Can Learn From Sneezing Ferrets and a Mutant Virus Debate","Partnership on AI","2020","report","medium.com","medium.com/partnership-on-ai/lessons-for-the-ai-community-from-the-h5n1-controversy-32432438a82e",0,"","policy"],["Conservatism in neocortex-like AGIs","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/c92YC89tznC7579Ej/conservatism-in-neocortex-like-agis",0,"",""],["Naturally Occurring Equivariance in Neural Networks","Chris Olah and 3 others","2020","report","Distill","distill.pub/2020/circuits/equivariance",0,"",""],["Idea: an AI governance group colocated with every AI research group!","capybaralet","2020","blog","EA Forum","forum.effectivealtruism.org/posts/PsHneZoLySmZ7W9dC/idea-an-ai-governance-group-colocated-with-every-ai-research",0,"","governance"],["Launching the Forecasting AI Progress Tournament","Tamay","2020","blog","LessWrong","www.lesswrong.com/posts/kWP5vcW4FSpM3is3J/launching-the-forecasting-ai-progress-tournament",0,"","forecasting"],["Traversing a Cognition Space","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/FNyqL7mxSkgLpck4w/traversing-a-cognition-space",0,"",""],["Minimal Maps, Semi-Decisions, and Neural Representations","Zachary Robertson","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ur4yr6WRCmEb5YfuH/minimal-maps-semi-decisions-and-neural-representations",0,"",""],["AI Problems Shared by Non-AI Systems","VojtaKovarik","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/iGs2jHc6Mcm3jtefk/ai-problems-shared-by-non-ai-systems",0,"",""],["AI Winter Is Coming - How to profit from it?","anonymous","2020","blog","LessWrong","www.lesswrong.com/posts/bqFu8fxokJSPjidJo/ai-winter-is-coming-how-to-profit-from-it",0,"","forecasting"],["Parsing by counterfactual","janus","2020","blog","generative.ink","generative.ink/posts/parsing-by-counterfactual/",0,"",""],["The AI Safety Game (UPDATED)","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/Nex8EgEJPsn7dvoQB/the-ai-safety-game-updated",0,"",""],["Values Form a Shifting Landscape (and why you might care)","VojtaKovarik","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4Qd2pDWeFPgYZfkSg/values-form-a-shifting-landscape-and-why-you-might-care",0,"",""],["An overview of 11 proposals for building safe advanced AI","Evan Hubinger","2020","paper","arXiv preprint","arxiv.org/abs/2012.07532",0,"","scalable-oversight debate evals"],["Learning in two-player games between transparent opponents","Adrian Hutter","2020","paper","arXiv preprint","arxiv.org/abs/2012.02671",0,"","interpretability agents"],["LessWrong is now a book, available for pre-order!","jacobjacob and Ben Pace","2020","blog","EA Forum","forum.effectivealtruism.org/posts/C838HYGS2a6wbkRTy/lesswrong-is-now-a-book-available-for-pre-order",0,"",""],["Long-Term Future Fund: Ask Us Anything!","AdamGleave","2020","blog","EA Forum","forum.effectivealtruism.org/posts/nsoFyaasfQipyiWzN/long-term-future-fund-ask-us-anything",0,"",""],["Understanding meta-trained algorithms through a Bayesian lens","DeepMind Safety Research","2020","blog","deepmindsafetyresearch.medium.com","deepmindsafetyresearch.medium.com/understanding-meta-trained-algorithms-through-a-bayesian-lens-5042a1acc1c2",0,"",""],["[AN #127]: Rethinking agency: Cartesian frames as a formalization of ways to carve up the world into an agent and its environment","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZZDHoqpHmChxEYMme/an-127-rethinking-agency-cartesian-frames-as-a-formalization",0,"","agents"],["Centre for the Study of Existential Risk Four Month Report June - September 2020","HaydnBelfield","2020","blog","EA Forum","forum.effectivealtruism.org/posts/EArLfuDz34zJHJZJx/centre-for-the-study-of-existential-risk-four-month-report-1",0,"","governance policy"],["Recursive Quantilizers II","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YNuJjRuxsWWzfvder/recursive-quantilizers-ii",0,"","scalable-oversight goodharts-law evals robustness"],["Value Alignment Verification","Daniel S. Brown and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.01557",0,"","evals agents"],["Aligning AI Optimization to Community Well-Being","Jonathan Stray","2020","report","link.springer.com","link.springer.com/10.1007/s42413-020-00086-3",0,"",""],["Fast reinforcement learning with generalized policy updates","André Barreto and 4 others","2020","report","pnas.org","www.pnas.org/lookup/doi/10.1073/pnas.1907370117",0,"","policy"],["Hacking AI","Andrew Lohn","2020","report","cset.georgetown.edu","cset.georgetown.edu/publication/hacking-ai/",0,"",""],["In a multipolar scenario, how do people expect systems to be trained to interact with systems developed by other labs?","JesseClifton","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Pkthep47ukcrK3MNm/in-a-multipolar-scenario-how-do-people-expect-systems-to-be",0,"",""],["Reinforcement Learning in Newcomblike Environments","James Bell and 3 others","2020","report","proceedings.neurips.cc","proceedings.neurips.cc/paper/2021/file/b9ed18a301c9f3d183938c451fa183df-Paper.pdf",0,"",""],["Rob Miles on Why should I care about AI safety-by Jeremie Harris on the Towards Data Science Podcast-date 20201202","Rob Miles and Jeremie Harris","2020","report","drive.google.com","drive.google.com/file/d/1FJMt4m2g7PaQDEGUesSZCDqcrydSjSEi/view?usp=share_link",0,"",""],["Sharing the World with Digital Minds","Aaron Gertler","2020","blog","EA Forum","forum.effectivealtruism.org/posts/4efXC5WZaHSHJJZTF/sharing-the-world-with-digital-minds",0,"",""],["Idealized Factored Cognition","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/S5oWwZMJBvfChSquW/idealized-factored-cognition",0,"",""],["November 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/11/30/november-2020-newsletter/",0,"",""],["Preface to the Sequence on Factored Cognition","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fnrpxdnodQmanibmB/preface-to-the-sequence-on-factored-cognition",0,"",""],["Is this a good way to bet on short timelines?","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/WiQRpz8M4hKQBQa3B/is-this-a-good-way-to-bet-on-short-timelines",0,"","forecasting robustness"],["Is this a good way to bet on short timelines?","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/kYa4dHP5MDnqmav2w/is-this-a-good-way-to-bet-on-short-timelines",0,"","forecasting robustness"],["[AN #126]: Avoiding wireheading by decoupling action feedback from action effects","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ak8a6fbhXbdqH3FgD/an-126-avoiding-wireheading-by-decoupling-action-feedback",0,"","reward-hacking"],["Can we convince people to work on AI safety without convincing them about AGI happening this century?","BrianTan","2020","blog","EA Forum","forum.effectivealtruism.org/posts/7YoEy2fEPdarHxYkC/can-we-convince-people-to-work-on-ai-safety-without",0,"",""],["Delegated agents in practice: How companies might end up selling AI services that act on behalf of consumers and coalitions, and what this implies for safety research","Remmelt","2020","blog","EA Forum","forum.effectivealtruism.org/posts/rExHeXfikaAxdMiDv/delegated-agents-in-practice-how-companies-might-end-up",0,"","agents"],["Energy efficiency of monarch butterfly flight","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-monarch-butterfly-flight/",0,"",""],["Energy efficiency of wandering albatross flight","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-wandering-albatross-flight/",0,"",""],["European Strategy on AI: Are we truly fostering social good?","Francesca Foffano and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.12863",0,"","deception"],["Contract Scheduling With Predictions","Spyros Angelopoulos and Shahin Kamali","2020","paper","arXiv preprint","arxiv.org/abs/2011.12439",0,"","robustness"],["Critiques of the Agent Foundations agenda?","Jsevillamol","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3jqKmuG7zq2qQLSBT/critiques-of-the-agent-foundations-agenda",0,"","agents theory"],["Energy efficiency of paramotors","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-paramotors/",0,"",""],["The next AI winter will be due to energy costs","hippke","2020","blog","LessWrong","www.lesswrong.com/posts/N7KYWJPmyzB6bJSYT/the-next-ai-winter-will-be-due-to-energy-costs-1",0,"","forecasting"],["Commentary on AGI Safety from First Principles","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/oiuZjPfknKsSc5waC/commentary-on-agi-safety-from-first-principles",0,"",""],["Continuing the takeoffs debate","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Tpn2Fx9daLvj28kes/continuing-the-takeoffs-debate",0,"","forecasting"],["Syntax, semantics, and symbol grounding, simplified","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/joPoxBpZjLNx8MKaF/syntax-semantics-and-symbol-grounding-simplified",0,"",""],["Transforming Worlds: Automated Involutive MCMC for Open-Universe Probabilistic Models","George Matheos and 5 others","2020","report","openreview.net","openreview.net/forum?id=8Itm8dQnJRc",0,"",""],["AGI Predictions","Pablo","2020","blog","EA Forum","forum.effectivealtruism.org/posts/YTjnCJuj3taaB6WYk/agi-predictions",0,"","forecasting"],["BARS: Joint Search of Cell Topology and Layout for Accurate and Efficient Binary ARchitectures","Tianchen Zhao and 8 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.10804",0,"",""],["Emergent Road Rules In Multi-Agent Driving Environments","Avik Pal and 3 others","2020","paper","International Conference on Learning Representations, 2021","arxiv.org/abs/2011.10753",0,"","agents"],["Jaan Tallinn: Fireside chat (2020)","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/he8zLBmPiNX8mFnzr/jaan-tallinn-fireside-chat-2020",0,"","governance"],["Jeffrey Ding: Bringing techno-globalism back: a romantically realist reframing of the US-China tech relationship","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/wfc46sHzSNspcem3k/jeffrey-ding-bringing-techno-globalism-back-a-romantically",0,"","governance policy"],["Non-Obstruction: A Simple Concept Motivating Corrigibility","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xts5wm3akbemk4pDa/non-obstruction-a-simple-concept-motivating-corrigibility",0,"","agents policy"],["Tan Zhi Xuan: AI alignment, philosophical pluralism, and the relevance of non-Western philosophy","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/D6nmypgiiPfS42pub/tan-zhi-xuan-ai-alignment-philosophical-pluralism-and-the",0,"",""],["UDT might not pay a Counterfactual Mugger","winwonce","2020","blog","LessWrong","www.lesswrong.com/posts/9KWs3rfvjCeGeJGzy/udt-might-not-pay-a-counterfactual-mugger",0,"","theory"],["Assessing Generalization in Reward Learning: Intro and Background","Max Chiswick and 2 others","2020","report","towardsdatascience.com","towardsdatascience.com/assessing-generalization-in-reward-learning-intro-and-background-da6c99d9e48",0,"",""],["Hiding Complexity","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6zbRy3aADCsRmFcgv/hiding-complexity",0,"",""],["Persuasion Tools: AI takeover without AGI or agency?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qKvn7rxP2mzJbKfcA/persuasion-tools-ai-takeover-without-agi-or-agency",0,"",""],["Persuasion Tools: AI takeover without AGI or agency?","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/77mpkNmKPjtictgDG/persuasion-tools-ai-takeover-without-agi-or-agency",0,"","forecasting"],["Announcing AI Safety Support","Linda Linsefors","2020","blog","EA Forum","forum.effectivealtruism.org/posts/wpQ2qhF8Z6oonsaPX/announcing-ai-safety-support",0,"",""],["Inner Alignment in Salt-Starved Rats","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/wcNEXDHowiWkRxDNv/inner-alignment-in-salt-starved-rats",0,"","interpretability"],["Misalignment and misuse: whose values are manifest?","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/misalignment-and-misuse-whose-values-are-manifest/",0,"",""],["Notes on Prudence","David Gross","2020","blog","LessWrong","www.lesswrong.com/posts/LSzSFeZpwsJB4Nowu/notes-on-prudence",0,"","theory"],["Some AI research areas and their relevance to existential safety","Andrew_Critch","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hvGoYXi2kgnS3vxqb/some-ai-research-areas-and-their-relevance-to-existential-1",0,"","agents theory"],["A Prototypeness Hierarchy of Realities","Tamsin Leake","2020","blog","carado.moe","carado.moe/prototype-realities.html",0,"",""],["Energy efficiency of The Spirit of Butt’s Farm","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-the-spirit-of-butts-farm/",0,"",""],["Normativity","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/tCex9F9YptGMpk2sT/normativity",0,"",""],["Should we postpone AGI until we reach safety?","otto.barten","2020","blog","LessWrong","www.lesswrong.com/posts/CXaQj85r4LtafCBi8/should-we-postpone-agi-until-we-reach-safety",0,"","governance forecasting"],["The ethics of AI for the Routledge Encyclopedia of Philosophy","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/b4cLtSam97ZhdJGMG/the-ethics-of-ai-for-the-routledge-encyclopedia-of",0,"",""],["The Pointers Problem: Human Values Are A Function Of Humans' Latent Variables","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gQY6LrTWJNkTv8YJR/the-pointers-problem-human-values-are-a-function-of-humans",0,"",""],["Using Unity to Help Solve Intelligence","Simon Carter and 9 others","2020","blog","deepmind.com","www.deepmind.com/blog/using-unity-to-help-solve-intelligence",0,"",""],["AI transparency: a matter of reconciling design with critique","Tomasz Hollanek","2020","report","doi.org","doi.org/10.1007/s00146-020-01110-y",0,"","interpretability"],["Avoiding Tampering Incentives in Deep RL via Decoupled Approval","Jonathan Uesato and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.08827",0,"","reward-hacking evals agents policy robustness"],["Efficient Exploration of Reward Functions in Inverse Reinforcement Learning via Bayesian Optimization","Sreejith Balakrishnan and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.08541",0,"","policy"],["Preventing Repeated Real World AI Failures by Cataloging Incidents: The AI Incident Database","Sean McGregor","2020","paper","arXiv preprint","arxiv.org/abs/2011.08512",0,"",""],["REALab: An Embedded Perspective on Tampering","Ramana Kumar and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.08820",0,"","agents theory"],["Understanding RL Vision","Jacob Hilton and 4 others","2020","report","Distill","distill.pub/2020/understanding-rl-vision",0,"",""],["Was the industrial revolution a drastic departure from historic trends?","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/was-the-industrial-revolution-a-drastic-departure-from-historic-trends/",0,"",""],["Donating against Short Term AI risks","Jan-Willem","2020","blog","EA Forum","forum.effectivealtruism.org/posts/wsCXLzXWEPr5pwHWm/donating-against-short-term-ai-risks",0,"",""],["Extortion beats brinksmanship, but the audience matters","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ofx82Y9a4zcETfT6Q/extortion-beats-brinksmanship-but-the-audience-matters",0,"",""],["How Roodman's GWP model translates to TAI timelines","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/RuPKfSELEC2nXYX57/how-roodman-s-gwp-model-translates-to-tai-timelines",0,"","governance forecasting"],["How Roodman's GWP model translates to TAI timelines","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/L23FgmpjsTebqcSZb/how-roodman-s-gwp-model-translates-to-tai-timelines",0,"","forecasting"],["A guide to Iterated Amplification & Debate","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/vhfATmAoJcN8RqGg6/a-guide-to-iterated-amplification-and-debate",0,"","scalable-oversight"],["Two Principles For Topia","Tamsin Leake","2020","blog","carado.moe","carado.moe/two-principles-for-topia.html",0,"",""],["Early Thoughts on Ontology/Grounding Problems","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/an29DQgYaKbyQprns/early-thoughts-on-ontology-grounding-problems",0,"",""],["A Self-Embedded Probabilistic Model","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SN8wFZsZiyBygc27k/a-self-embedded-probabilistic-model",0,"",""],["Active Reinforcement Learning: Observing Rewards at a Cost","David Krueger and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.06709",0,"","evals agents"],["Misalignment and misuse: whose values are manifest?","KatjaGrace","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/AomSXpFcqmgeDyWWo/misalignment-and-misuse-whose-values-are-manifest",0,"",""],["Communication Prior as Alignment Strategy","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zAvhvGa6ToieNGuy2/communication-prior-as-alignment-strategy",0,"",""],["fiction about AI risk","Ann Garth","2020","blog","EA Forum","forum.effectivealtruism.org/posts/5aM8qQE3Pq9D8HxrR/fiction-about-ai-risk",0,"",""],["Learning Latent Representations to Influence Multi-Agent Interaction","Annie Xie and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.06619",0,"","agents policy"],["Performance of Bounded-Rational Agents With the Ability to Self-Modify","Jakub Tětek and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.06275",0,"","agents"],["[AN #125]: Neural network scaling laws across multiple modalities","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XPqMbtpbku8aN55wd/an-125-neural-network-scaling-laws-across-multiple",0,"","scaling-laws"],["A Correspondence Theorem in the Maximum Entropy Framework","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XMGWdfTC7XjgTz3X7/a-correspondence-theorem-in-the-maximum-entropy-framework",0,"",""],["CHAI Internship Application","martinfukui","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TgENS48GLJfei7FJu/chai-internship-application",0,"",""],["Fooling the primate brain with minimal, targeted image manipulation","Li Yuan and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.05623",0,"","robustness"],["I Know What You Meant: Learning Human Objectives by (Under)estimating Their Choice Set","Ananth Jonnavittula and Dylan P. Losey","2020","paper","arXiv preprint","arxiv.org/abs/2011.06118",0,"","robustness"],["Learning Normativity: A Research Agenda","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2JGu9yxiJkoGdQR4s/learning-normativity-a-research-agenda",0,"","rlhf robustness"],["Time in Cartesian Frames","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/JTzLjARpevuNpGPZm/time-in-cartesian-frames",0,"","theory"],["Eight Definitions of Observability","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/5R9dRqTREZriN9iL7/eight-definitions-of-observability",0,"","agents theory"],["Energy efficiency of MacCready Gossamer Albatross","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/maccready-gossamer-albatross/",0,"",""],["It Takes a Village: The Shared Responsibility of 'Raising' an Autonomous Weapon","Amritha Jayanti and Shahar Avin","2020","report","cser.ac.uk","www.cser.ac.uk/media/uploads/files/It_Takes_a_Village__The_Shared_Responsibility_of_Raising_an_Autonomous_Weapon.pdf",0,"",""],["Natural Language Inference in Context -- Investigating Contextual Reasoning over Long Texts","Hanmeng Liu and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.04864",0,"","benchmarks"],["What Did You Think Would Happen? Explaining Agent Behaviour Through Intended Outcomes","Herman Yau and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.05064",0,"","agents"],["A Theory of Universal Learning","Olivier Bousquet and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.04483",0,"",""],["Clarifying inner alignment terminology","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SzecSPYxqRa5GCaSF/clarifying-inner-alignment-terminology",0,"","robustness"],["Committing, Assuming, Externalizing, and Internalizing","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/5HMqSGQ9ad9r9Hibw/committing-assuming-externalizing-and-internalizing",0,"","agents theory"],["Risk Assessment for Machine Learning Models","Paul Schwerdtner and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.04328",0,"","robustness theory"],["Why You Should Care About Goal-Directedness","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/q9BmNh35xgXPRgJhm/why-you-should-care-about-goal-directedness",0,"",""],["Cartesian Frames Definitions","Rob Bensinger","2020","blog","LessWrong","www.lesswrong.com/posts/kLLu387fiwbis3otQ/cartesian-frames-definitions",0,"","theory"],["When Hindsight Isn't 20/20: Incentive Design With Imperfect Credit Allocation","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XPRAY34Sutc2wWYZf/when-hindsight-isn-t-20-20-incentive-design-with-imperfect",0,"",""],["How can I bet on short timelines?","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/DDTYxpK42B495MPqM/how-can-i-bet-on-short-timelines",0,"","forecasting"],["How can I bet on short timelines?","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/4FhiSuNv4QbtKDzL8/how-can-i-bet-on-short-timelines",0,"","forecasting"],["the scaling “inconsistency”: openAI’s new insight","nostalgebraist","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/diutNaWF669WgEt3v/the-scaling-inconsistency-openai-s-new-insight",0,"",""],["Additive and Multiplicative Subagents","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hxbjEgjNTSbdXqDFE/additive-and-multiplicative-subagents",0,"","agents theory"],["Does SGD Produce Deceptive Alignment?","Mark Xu","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ocWqg2Pf2br4jMmKA/does-sgd-produce-deceptive-alignment",0,"","alignment-faking deception"],["Energy efficiency of Airbus A320","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-airbus-a320/",0,"",""],["Energy efficiency of Boeing 747-400","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-boeing-747-400/",0,"",""],["Energy efficiency of North American P-51 Mustang","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-north-american-p-51-mustang/",0,"",""],["Consider paying me to do AI safety research work","Rupert","2020","blog","EA Forum","forum.effectivealtruism.org/posts/6oGp7XdGySzAGq4QC/consider-paying-me-to-do-ai-safety-research-work",0,"",""],["Defining capability and alignment in gradient descent","Edouard Harris","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xg2YycEfCnLYrCcjy/defining-capability-and-alignment-in-gradient-descent",0,"","robustness"],["Energy efficiency of Vickers Vimy plane","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-vickers-vimy-plane/",0,"",""],["Energy efficiency of Wright model B","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-wright-model-b/",0,"",""],["Sub-Sums and Sub-Tensors","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LAHXvi4qwXogmdTHd/sub-sums-and-sub-tensors-1",0,"","theory"],["What considerations influence whether I have more influence over short or long timelines?","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/xqQe85ZEs8KHxAbaF/what-considerations-influence-whether-i-have-more-influence",0,"","forecasting"],["What considerations influence whether I have more influence over short or long timelines?","Daniel Kokotajlo","2020","blog","LessWrong","www.lesswrong.com/posts/pTK2cDnXBB5tpoP74/what-considerations-influence-whether-i-have-more-influence",0,"","forecasting"],["[AN #124]: Provably safe exploration through shielding","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TJqfcEyDdLwkDxZZC/an-124-provably-safe-exploration-through-shielding",0,"",""],["Energy efficiency of Wright Flyer","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/energy-efficiency-of-wright-flyer/",0,"",""],["Multiplicative Operations on Cartesian Frames","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/srTD8DgTCR27udzoe/multiplicative-operations-on-cartesian-frames",0,"","theory"],["Confucianism in AI Alignment","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3aDeaJzxinoGNWNpC/confucianism-in-ai-alignment",0,"","agents"],["Subagents of Cartesian Frames","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/nwrkwTd6uKBesYYfx/subagents-of-cartesian-frames",0,"","agents theory"],["Ask Your Humans: Using Human Instructions to Improve Generalization in Reinforcement Learning","Valerie Chen and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.00517",0,"","agents robustness"],["Automated intelligence is not AI","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/automated-intelligence-is-not-ai/",0,"",""],["\"Inner Alignment Failures\" Which Are Actually Outer Alignment Failures","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/HYERofGZE6j9Tuigi/inner-alignment-failures-which-are-actually-outer-alignment",0,"",""],["Containing the AI... Inside a Simulated Reality","HumaneAutomation","2020","blog","LessWrong","www.lesswrong.com/posts/J5j3wypPgcLyrKmwZ/containing-the-ai-inside-a-simulated-reality",0,"",""],["Amplifying GPT-3 on closed-ended questions","janus","2020","blog","generative.ink","generative.ink/posts/amplifying-gpt-3-on-closed-ended-questions/",0,"",""],["Functors and Coarse Worlds","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/GYQwJsChoRosjdW2r/functors-and-coarse-worlds",0,"","theory"],["Responses to Christiano on takeoff speeds?","Richard_Ngo","2020","blog","LessWrong","www.lesswrong.com/posts/sACaK4tBvPHkEQW9w/responses-to-christiano-on-takeoff-speeds",0,"","forecasting"],["AI risk hub in Singapore?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QTL5tRz7Q54bpcwdE/ai-risk-hub-in-singapore-1",0,"",""],["AI risk hub in Singapore?","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/926FtZiEERsGfPPv9/ai-risk-hub-in-singapore",0,"",""],["Controllables and Observables, Revisited","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/z3S2xnoDYfohrQQoe/controllables-and-observables-revisited",0,"","theory"],["Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","Brijen Thananjeyan and 9 others","2020","paper","Robotics and Automation Letters (RA-L) and International\n  Conference on Robotics and Automation (ICRA) 2021","arxiv.org/abs/2010.15920",0,"","evals agents policy"],["[AN #123]: Inferring what is valuable in order to align recommender systems","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/HbtRFDiyTDpPfRLqm/an-123-inferring-what-is-valuable-in-order-to-align",0,"",""],["Biextensional Equivalence","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pWruFSY7494vnucCE/biextensional-equivalence",0,"","theory"],["Draft papers for REALab and Decoupled Approval on tampering","Jonathan Uesato and Ramana Kumar","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/X23q6T4CDifHykqi4/draft-papers-for-realab-and-decoupled-approval-on-tampering",0,"","reward-hacking theory"],["Scaling Laws for Autoregressive Generative Modeling","Tom Henighan and 18 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.14701",0,"","training-data scaling-laws"],["Dutch-Booking CDT: Revised Argument","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/X7k23zk9aBjjpgLd3/dutch-booking-cdt-revised-argument",0,"","theory"],["Generative Temporal Difference Learning for Infinite-Horizon Prediction","Michael Janner and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.14496",0,"","evals agents policy"],["Learning to be Safe: Deep RL with a Safety Critic","Krishnan Srinivasan and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.14603",0,"","policy"],["Rohin Shah - Effective altruism, AI safety, and learning human preferences from the world_s state-by Towards Data Science-video_id uHiL6GNXHvw-date 20201028","Rohin Shah and Jeremie Harris","2020","report","drive.google.com","drive.google.com/file/d/1ZUN1YAJsy9aq1F-oXKzM43GlvvnwyI7d/view?usp=share_link",0,"",""],["Security Mindset and Takeoff Speeds","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Lfk2FXBwrpoM6Jm7p/security-mindset-and-takeoff-speeds",0,"","forecasting"],["A Correspondence Theorem","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/FWuByzM9T5qq2PF2n/a-correspondence-theorem",0,"",""],["Additive Operations on Cartesian Frames","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ewkYgtZapQRtDPT2F/additive-operations-on-cartesian-frames",0,"","theory"],["Supervised learning of outputs in the brain","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jNrDzyc8PJ9HXtGFm/supervised-learning-of-outputs-in-the-brain",0,"",""],["Time for AI to cross the human range in English draughts","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/time-for-ai-to-cross-the-human-range-in-english-draughts/",0,"",""],["4 Years Later: President Trump and Global Catastrophic Risk","HaydnBelfield","2020","blog","EA Forum","forum.effectivealtruism.org/posts/6CRvK76onGdHTqYoK/4-years-later-president-trump-and-global-catastrophic-risk",0,"","governance"],["Artificial intelligence career stories","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/KwcJ8MfyyB2mP4rwa/artificial-intelligence-career-stories",0,"",""],["Buck Shlegeris: How I think students should orient to AI safety","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/9CtcDEZCAgNkJF9pf/buck-shlegeris-how-i-think-students-should-orient-to-ai",0,"",""],["How to build a safe advanced AI (Evan Hubinger) | What's up in AI safety? (Asya Bergal)","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/5nteR48KqgGCpuatX/how-to-build-a-safe-advanced-ai-evan-hubinger-or-what-s-up",0,"",""],["Reply to Jebari and Lundborg on Artificial Superintelligence","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/rokpjK3jcy5aKKwiT/reply-to-jebari-and-lundborg-on-artificial-superintelligence-1",0,"",""],["Exemplary natural images explain CNN activations better than feature visualizations","Judy Borowski and Roland S. Zimmermann","2020","paper","arXiv preprint","arxiv.org/abs/2010.12606",0,"","interpretability mechanistic-interpretability"],["Humans are stunningly rational and stunningly irrational","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/X8KQBjszbSDXzBwgP/humans-are-stunningly-rational-and-stunningly-irrational",0,"",""],["October 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/10/23/october-2020-newsletter/",0,"",""],["An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","Alexey Dosovitskiy and 11 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.11929",0,"","benchmarks"],["Enabling certification of verification-agnostic networks via memory-efficient semidefinite programming","Sumanth Dathathri and 10 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.11645",0,"","assurance"],["Exploring the Nuances of Designing (with/for) Artificial Intelligence","Niya Stoimenova and Rebecca Price","2020","paper","Design Issues, 36(4), 45-55 (2020)","arxiv.org/abs/2010.15578",0,"","evals deception"],["Introduction to Cartesian Frames","Scott Garrabrant","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BSpdshJWGAW6TuNzZ/introduction-to-cartesian-frames",0,"","theory"],["The date of AI Takeover is not the day the AI takes over","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/JPan54R525D68NoEt/the-date-of-ai-takeover-is-not-the-day-the-ai-takes-over",0,"","forecasting"],["[AN #122]: Arguing for AGI-driven existential risk from first principles","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/MxHiYZJjYm53ATxhb/an-122-arguing-for-agi-driven-existential-risk-from-first",0,"",""],["AGI safety from first principles","richard_ngo","2020","blog","EA Forum","forum.effectivealtruism.org/posts/MMtbCDTNP3M53N3Dc/agi-safety-from-first-principles",0,"",""],["Box inversion hypothesis","Jan Kulveit","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TQwXPHfyyQwr22NMh/box-inversion-hypothesis",0,"",""],["Problems Involving Abstraction?","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XArPqdkwCtEekgYxv/problems-involving-abstraction",0,"",""],["Robust Imitation Learning from Noisy Demonstrations","Voot Tangkaratt and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.10181",0,"","agents policy training-data"],["Time for AI to cross the human range in StarCraft","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/time-for-ai-to-cross-the-human-range-in-starcraft/",0,"",""],["Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","Alessandro Giuseppi and Antonio Pietrabissa","2020","paper","IEEE Control Systems Letters, vol. 4, no. 3, pp. 755-760, July\n  2020","arxiv.org/abs/2010.09468",0,"",""],["Future Indices","Michael Page and 2 others","2020","report","cset.georgetown.edu","cset.georgetown.edu/publication/future-indices/",0,"",""],["RobustBench: a standardized adversarial robustness benchmark","Francesco Croce and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.09670",0,"","evals benchmarks robustness"],["Time for AI to cross the human performance range in ImageNet image classification","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/time-for-ai-to-cross-the-human-performance-range-in-imagenet-image-classification/",0,"",""],["HITL thought experiment","janus","2020","blog","generative.ink","generative.ink/posts/hitl-thought-experiment/",0,"",""],["Time for AI to cross the human performance range in Go","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/time-for-ai-to-cross-the-human-performance-range-in-go/",0,"",""],["Avoiding Side Effects By Considering Future Tasks","Victoria Krakovna and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.07877",0,"","agents policy"],["Do's and Don'ts for Human and Digital Worker Integration","Vinod Muthusamy and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.07738",0,"","evals"],["The case for taking AI seriously as a threat to humanity (Kelsey Piper)","EA Handbook","2020","blog","EA Forum","forum.effectivealtruism.org/posts/94pRmwWAqmhdA82CF/the-case-for-taking-ai-seriously-as-a-threat-to-humanity-1",0,"",""],["Time for AI to cross the human performance range in chess","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/time-for-ai-to-cross-the-human-performance-range-in-chess/",0,"",""],["[AN #121]: Forecasting transformative AI timelines using biological anchors","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/cxQtz3RP4qsqTkEwL/an-121-forecasting-transformative-ai-timelines-using",0,"","forecasting"],["The Colliding Exponentials of AI","Vermillion","2020","blog","LessWrong","www.lesswrong.com/posts/QWuegBA9kGBv3xBFy/the-colliding-exponentials-of-ai",0,"","forecasting"],["The Solomonoff Prior is Malign","Mark Xu","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Tr7tAyt5zZpdTwTQK/the-solomonoff-prior-is-malign",0,"",""],["Knowledge, manipulation, and free will","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2dKvTYYN4PTT7g4of/knowledge-manipulation-and-free-will",0,"",""],["Longtermist reasons to work for innovative governments","ac","2020","blog","EA Forum","forum.effectivealtruism.org/posts/EgAGoFazXe9yPbjcm/longtermist-reasons-to-work-for-innovative-governments",0,"","governance policy"],["The Achilles Heel Hypothesis for AI","scasper","2020","blog","LessWrong","www.lesswrong.com/posts/o7eWu5Gzd82dw9dJS/the-achilles-heel-hypothesis-for-ai",0,"","robustness theory"],["Toy Problem: Detective Story Alignment","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4kYkYSKSALH4JaQ99/toy-problem-detective-story-alignment",0,"","debate robustness"],["TIO: A mental health chatbot","Sanjay","2020","blog","EA Forum","forum.effectivealtruism.org/posts/yWGaezWTuPY6LcJ4f/tio-a-mental-health-chatbot",0,"",""],["Safe Reinforcement Learning with Natural Language Constraints","Tsung-Yen Yang and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.05150",0,"","benchmarks agents policy"],["Logical Foundations of Government Policy","FCCC","2020","blog","LessWrong","www.lesswrong.com/posts/xcFn7GGrypEFuDjmd/logical-foundations-of-government-policy",0,"","policy theory"],["If GPT-6 is human-level AGI but costs $200 per page of output, what would happen?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/kcWHfRnLMDDgsbJfd/if-gpt-6-is-human-level-agi-but-costs-usd200-per-page-of",0,"",""],["Parameterized Reinforcement Learning for Optical System Optimization","Heribert Wankerl and 4 others","2020","paper","J. Phys. D: Appl. Phys. 54 305104 (2021)","arxiv.org/abs/2010.05769",0,"","policy"],["[Link] How understanding valence could help make future AIs safer","Milan_Griffes","2020","blog","EA Forum","forum.effectivealtruism.org/posts/pTZ5uCA8memQ9faje/link-how-understanding-valence-could-help-make-future-ais",0,"",""],["Information-Driven Adaptive Sensing Based on Deep Reinforcement Learning","Abdulmajid Murad and 3 others","2020","paper","10th International Conference on the Internet of Things (IoT20),\n  October 6-9, 2020, Malmo, Sweden","arxiv.org/abs/2010.04112",0,"","monitoring"],["[AN #120]: Tracing the intellectual roots of AI and AI alignment","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/b9b4y2azGjthGBEFb/an-120-tracing-the-intellectual-roots-of-ai-and-ai-alignment",0,"",""],["A framework for predicting, interpreting, and improving Learning Outcomes","Chintan Donda and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.02629",0,"","evals deception agents policy"],["Providing Actionable Feedback in Hiring Marketplaces using Generative Adversarial Networks","Daniel Nemirovsky and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.02419",0,"","deception"],["Safety Aware Reinforcement Learning (SARL)","Santiago Miret and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.02846",0,"","scalable-oversight debate benchmarks agents policy"],["The Alignment Problem: Machine Learning and Human Values","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gYfgWSxCpFdk2cZfE/the-alignment-problem-machine-learning-and-human-values",0,"",""],["The Alignment Problem: Machine Learning and Human Values","Brian Christian","2020","report","goodreads.com","www.goodreads.com/book/show/50489349-the-alignment-problem",0,"",""],["Learning to Generalize for Sequential Decision Making","Xusen Yin and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.02229",0,"",""],["AGI safety from first principles: Conclusion","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ni8ocGupB2kGG2fA7/agi-safety-from-first-principles-conclusion",0,"",""],["AI race considerations in a report by the U.S. House Committee on Armed Services","NunoSempere","2020","blog","LessWrong","www.lesswrong.com/posts/87aqBTkhTgfzhu5po/ai-race-considerations-in-a-report-by-the-u-s-house",0,"","governance"],["Feedback Request on EA Philippines' Career Advice Research for Technical AI Safety","BrianTan","2020","blog","EA Forum","forum.effectivealtruism.org/posts/XkDXSmqoKhR7ezKYf/feedback-request-on-ea-philippines-career-advice-research",0,"",""],["Gender Bootstrappism","Tamsin Leake","2020","blog","carado.moe","carado.moe/gender-bootstrappism.html",0,"",""],["Real quick, on free will","Tamsin Leake","2020","blog","carado.moe","carado.moe/free-will.html",0,"",""],["Socialism as a conspiracy theory","Tamsin Leake","2020","blog","carado.moe","carado.moe/socialism-conspiracy.html",0,"",""],["Where next for piracy ?","Tamsin Leake","2020","blog","carado.moe","carado.moe/where-next-piracy.html",0,"",""],["Word Report #2","Tamsin Leake","2020","blog","carado.moe","carado.moe/word-report-2.html",0,"",""],["AGI safety from first principles: Control","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eGihD5jnD6LFzgDZA/agi-safety-from-first-principles-control",0,"",""],["AGI safety from first principles: Alignment","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PvA2gFMAaHCHfMXrw/agi-safety-from-first-principles-alignment",0,"",""],["Economic growth under transformative AI","Phillip Trammell and Anton Korinek","2020","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/philip-trammell-and-anton-korinek-economic-growth-under-transformative-ai/",0,"",""],["Emergent Social Learning via Multi-agent Reinforcement Learning","Kamal Ndousse and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.00581",0,"","agents"],["Hiring engineers and researchers to help align GPT-3","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dJQo7xPn4TyGnKgeC/hiring-engineers-and-researchers-to-help-align-gpt-3",0,"","rlhf"],["Hiring engineers and researchers to help align GPT-3","Paul_Christiano","2020","blog","EA Forum","forum.effectivealtruism.org/posts/dZTWQQash9tjy9AwH/hiring-engineers-and-researchers-to-help-align-gpt-3",0,"",""],["Mediating Artificial Intelligence Developments through Negative and Positive Incentives","The Anh Han and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.00403",0,"","governance policy"],["Quantifying the probability of existential catastrophe: A reply to Beard et al.","Seth D. Baum","2020","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0016328720300987",0,"",""],["\"Zero Sum\" is a misnomer.","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/D8ds9idKWbwzCseCh/zero-sum-is-a-misnomer",0,"",""],["[AN #119]: AI safety when agents are shaped by environments, not rewards","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kx7nv8dHtFig9ud7C/an-119-ai-safety-when-agents-are-shaped-by-environments-not",0,"","agents"],["Competence vs Alignment","Ariel Kwiatkowski","2020","blog","LessWrong","www.lesswrong.com/posts/9GC35E9JkkcLtBi7Y/competence-vs-alignment",0,"",""],["Learning Rewards from Linguistic Feedback","Theodore R. Sumers and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.14715",0,"","rlhf evals agents"],["“Unsupervised” translation as an (intent) alignment problem","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/saRRRdMnMPXXtQBNi/unsupervised-translation-as-an-intent-alignment-problem",0,"",""],["“Unsupervised” translation as an (intent) alignment problem","Paul Christiano","2020","report","ai-alignment.com","ai-alignment.com/unsupervised-translation-as-a-safety-problem-99ae1f9b6b68",0,"",""],["AGI safety from first principles: Goals and Agency","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bz5GdmCWj8o48726N/agi-safety-from-first-principles-goals-and-agency",0,"",""],["Learning to Play Against Any Mixture of Opponents","Max Olan Smith and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.14180",0,"",""],["Trust-Region Method with Deep Reinforcement Learning in Analog Design Space Exploration","Kai-En Yang and 8 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.13772",0,"","mechanistic-interpretability agents"],["AGI safety from first principles: Introduction","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8xRSjC76HasLnMGSf/agi-safety-from-first-principles-introduction",0,"",""],["AGI safety from first principles: Superintelligence","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eG3WhHS8CLNxuH6rT/agi-safety-from-first-principles-superintelligence",0,"",""],["Benefits of Assistance over Reward Learning","Rohin Shah and 9 others","2020","report","openreview.net","openreview.net/forum?id=DFIoGDZejIB",0,"",""],["The EMPATHIC Framework for Task Learning from Implicit Human Feedback","Yuchen Cui and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.13649",0,"","rlhf agents"],["The Grey Hoodie Project: Big Tobacco, Big Tech, and the threat on academic integrity","Mohamed Abdalla and Moustafa Abdalla","2020","paper","arXiv preprint","arxiv.org/abs/2009.13676",0,"",""],["What Decision Theory is Implied By Predictive Processing?","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8Ziz5BQjtuhr9orm4/what-decision-theory-is-implied-by-predictive-processing",0,"","theory"],["The whirlpool of reality","Gordon Seidoh Worley","2020","blog","LessWrong","www.lesswrong.com/posts/Zw5STvhmGNzuQYM5B/the-whirlpool-of-reality",0,"","theory"],["What to do with imitation humans, other than asking them what the right thing to do is?","Charlie Steiner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LAR2ajpFDueNg45Mk/what-to-do-with-imitation-humans-other-than-asking-them-what",0,"",""],["Inverse Rational Control with Partially Observable Continuous Nonlinear Dynamics","Minhae Kwon and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.12576",0,"","agents policy robustness"],["Neurosymbolic Reinforcement Learning with Formally Verified Exploration","Greg Anderson and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.12612",0,"","policy"],["Examples of self-governance to reduce technology risk?","jia","2020","blog","EA Forum","forum.effectivealtruism.org/posts/KJw6RDm4M6gAfqW6X/examples-of-self-governance-to-reduce-technology-risk",0,"","governance"],["[AN #118]: Risks, solutions, and prioritization in a world with many AI systems","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8eX8DJctsACtR2sfX/an-118-risks-solutions-and-prioritization-in-a-world-with",0,"",""],["Dehumanisation *errors*","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SfNwpyL7o49ohYyWB/dehumanisation-errors",0,"",""],["A narrowing of AI research?","Joel Klinger and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.10385",0,"",""],["Anthropomorphisation vs value learning: type 1 vs type 2 errors","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LkytHQSKbQFf6toW5/anthropomorphisation-vs-value-learning-type-1-vs-type-2",0,"",""],["AMA: Markus Anderljung (PM at GovAI, FHI)","MarkusAnderljung","2020","blog","EA Forum","forum.effectivealtruism.org/posts/6h3a9bvJ2uYBfWxEM/ama-markus-anderljung-pm-at-govai-fhi-1",0,"","governance"],["Needed: AI infohazard policy","Vanessa Kosoy","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3D3DsX5rMbk3jEZ5h/needed-ai-infohazard-policy",0,"","policy"],["Clarifying “What failure looks like”","Sam Clarke","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/v6Q7T335KCMxujhZu/clarifying-what-failure-looks-like",0,"",""],["Hidden Incentives for Auto-Induced Distributional Shift","David Krueger and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.09153",0,"",""],["Humans learn too: Better Human-AI Interaction using Optimized Human Inputs","Johannes Schneider","2020","paper","arXiv preprint","arxiv.org/abs/2009.09266",0,"","robustness"],["Why GPT wants to mesa-optimize & how we might change this","John_Maxwell","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BGD5J2KAoNmpPMzMQ/why-gpt-wants-to-mesa-optimize-and-how-we-might-change-this",0,"",""],["Draft report on AI timelines","Ajeya Cotra","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/KrJfoZzpSDpnrv9va/draft-report-on-ai-timelines",0,"","forecasting"],["Efficient Reinforcement Learning Development with RLzoo","Zihan Ding and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.08644",0,"","evals agents"],["Enterprise AI Canvas -- Integrating Artificial Intelligence into Business","U. Kerzel","2020","paper","arXiv preprint","arxiv.org/abs/2009.11190",0,"","evals deception"],["The \"Backchaining to Local Search\" Technique in AI Alignment","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qEjh8rpxjG4qGtfuK/the-backchaining-to-local-search-technique-in-ai-alignment",0,"",""],["AI Governance: Opportunity and Theory of Impact","Allan Dafoe","2020","blog","EA Forum","forum.effectivealtruism.org/posts/42reWndoTEhFqu6T8/ai-governance-opportunity-and-theory-of-impact",0,"","governance policy"],["Alignment for Advanced Machine Learning Systems","Jessica Taylor and 7 others","2020","report","oxford.universitypressscholarship.com","oxford.universitypressscholarship.com/view/10.1093/oso/9780190905033.001.0001/oso-9780190905033-chapter-13",0,"",""],["Distributional Generalization: A New Kind of Generalization","Preetum Nakkiran and Yamini Bansal","2020","paper","arXiv preprint","arxiv.org/abs/2009.08092",0,"",""],["Learnable Strategies for Bilateral Agent Negotiation over Multiple Issues","Pallavi Bagga and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.08302",0,"","interpretability evals agents"],["[AN #117]: How neural nets would fare under the TEVV framework","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8H5JbowLTJoNHzLuH/an-117-how-neural-nets-would-fare-under-the-tevv-framework",0,"",""],["Applying the Counterfactual Prisoner's Dilemma to Logical Uncertainty","Chris_Leong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XzvR3QKkt9EPbAYyT/applying-the-counterfactual-prisoner-s-dilemma-to-logical",0,"","deception theory"],["Are social media algorithms an existential risk?","Barry Grimes","2020","blog","EA Forum","forum.effectivealtruism.org/posts/E4gfMSqmznDwMrv9q/are-social-media-algorithms-an-existential-risk",0,"",""],["New report on how much computational power it takes to match the human brain (Open Philanthropy)","Aaron Gertler","2020","blog","EA Forum","forum.effectivealtruism.org/posts/nGQJEYp5X2pCbeweg/new-report-on-how-much-computational-power-it-takes-to-match",0,"","forecasting"],["Comparing Utilities","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/cYsGrWEzjb324Zpjx/comparing-utilities",0,"","agents"],["My computational framework for the brain","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/diruo47z32eprenTg/my-computational-framework-for-the-brain",0,"",""],["Decision Theory is multifaceted","Michele Campolo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/vrJBQZJpvswXFFkcd/decision-theory-is-multifaceted",0,"","theory"],["Egan's Theorem?","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/74crqQnH8v9JtJcda/egan-s-theorem",0,"",""],["Towards the Quantification of Safety Risks in Deep Neural Networks","Peipei Xu and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.06114",0,"","evals benchmarks robustness"],["How Much Computational Power Does It Take to Match the Human Brain?","habryka","2020","blog","LessWrong","www.lesswrong.com/posts/pgQ3m73kpjGDgKuRM/how-much-computational-power-does-it-take-to-match-the-human",0,"","forecasting"],["An Argumentation-based Approach for Identifying and Dealing with Incompatibilities among Procedural Goals","Mariela Morveli-Espinoza and 4 others","2020","paper","International Journal of Approximate Reasoning, year 2019, vol.\n  105, pp. 1-26","arxiv.org/abs/2009.05186",0,"","deception agents"],["Communicating with Interactive Articles","Fred Hohman and 3 others","2020","report","Distill","distill.pub/2020/communicating-with-interactive-articles",0,"",""],["How Much Computational Power Does It Take to Match the Human Brain?","Joseph Carlsmith","2020","report","openphilanthropy.org","www.openphilanthropy.org/brain-computation-report",0,"",""],["September 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/09/10/september-2020-newsletter/",0,"",""],["The AIQ Meta-Testbed: Pragmatically Bridging Academic AI Testing and Industrial Q Needs","Markus Borg","2020","paper","arXiv preprint","arxiv.org/abs/2009.05260",0,"","assurance"],["DeepSpeed: Extreme-scale model training for everyone","DeepSpeed Team and Rangan Majumder","2020","report","microsoft.com","www.microsoft.com/en-us/research/blog/deepspeed-extreme-scale-model-training-for-everyone/",0,"",""],["Do mesa-optimizer risk arguments rely on the train-test paradigm?","Ben Cottier","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/j5foHZhZ7RBhwRL7Z/do-mesa-optimizer-risk-arguments-rely-on-the-train-test",0,"",""],["Importance Weighted Policy Learning and Adaptation","Alexandre Galashov and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.04875",0,"","policy"],["Measurement in AI Policy: Opportunities and Challenges","Saurabh Mishra and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.09071",0,"","benchmarks policy robustness"],["Safety via selection for obedience","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/7jNveWML34EsjCD4c/safety-via-selection-for-obedience",0,"",""],["[AN #116]: How to make explanations of neurons compositional","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jCZhy3nqH2MoethZQ/an-116-how-to-make-explanations-of-neurons-compositional",0,"",""],["Beneficial and Harmful Explanatory Machine Learning","Lun Ai and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.06410",0,"",""],["Safer sandboxing via collective separation","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Fji2nHBaB6SjdSscr/safer-sandboxing-via-collective-separation",0,"",""],["Determining core values & existential self-determination","Tamsin Leake","2020","blog","carado.moe","carado.moe/core-vals-exist-selfdet.html",0,"",""],["(The Cartoon Guide to) Lob’s Theorem","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/lobs-theorem",0,"",""],["5-Minute Singularity Intro","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/intro",0,"",""],["A Technical Explanation of Technical Explanation","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/technical",0,"",""],["An Intuitive Explanation of Bayes’ Theorem","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/bayes",0,"",""],["Artifacts","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/other/artifacts",0,"",""],["Artificial Intelligence as a Positive and Negative Factor in Global Risk","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/ai-risk",0,"",""],["Cognitive Biases Potentially Affecting Judgment of Global Risks","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/cognitive-biases",0,"",""],["Dark Lord’s Answer","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/other/fiction/dark-lords-answer",0,"",""],["Girl Intercorrupted","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/other/fiction/girl-intercorrupted",0,"",""],["Overcoming Bias","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/overcoming-bias",0,"",""],["Prospiracy Theory","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/other/fiction/prospiracy-theory",0,"",""],["Singularity Fun Theory","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/fun-theory",0,"",""],["The AI-Box Experiment:","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/aibox",0,"",""],["The Power of Intelligence","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/power",0,"",""],["The Simple Truth","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/rational/the-simple-truth",0,"",""],["Three Major Singularity Schools","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/schools",0,"",""],["Transhumanism as Simplified Humanism","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/singularity/simplified",0,"",""],["Yehuda Yudkowsky, 1985-2004","Eliezer S. Yudkowsky","2020","blog","yudkowsky.net","www.yudkowsky.net/other/yehuda",0,"",""],["Using GPT-N to Solve Interpretability of Neural Networks: A Research Agenda","Logan Riggs and Gurkenglas","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zXfqftW8y69YzoXLj/using-gpt-n-to-solve-interpretability-of-neural-networks-a",0,"","interpretability"],["[AN #115]: AI safety research problems in the AI-GA framework","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bevquxoYwkMx3NK6L/an-115-ai-safety-research-problems-in-the-ai-ga-framework",0,"",""],["Estimating the Brittleness of AI: Safety Integrity Levels and the Need for Testing Out-Of-Distribution Performance","Andrew J. Lohn","2020","paper","arXiv preprint","arxiv.org/abs/2009.00802",0,"","evals policy robustness"],["Are we living at the hinge of history","William MacAskill","2020","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/wp-content/uploads/William-MacAskill_Are-we-living-at-the-hinge-of-history.pdf",0,"",""],["In defence of fanaticism","Hayden Wilkinson","2020","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/wp-content/uploads/Hayden-Wilkinson_In-defence-of-fanaticism.pdf",0,"",""],["(Humor) AI Alignment Critical Failure Table","Kaj_Sotala","2020","blog","LessWrong","www.lesswrong.com/posts/DW8CjebNuzcvYHBXv/humor-ai-alignment-critical-failure-table",0,"",""],["A course for the general public on AI","LeandroD","2020","blog","EA Forum","forum.effectivealtruism.org/posts/pT9QTeAGT4GmMbT5w/a-course-for-the-general-public-on-ai",0,"","governance"],["interpreting GPT: the logit lens","nostalgebraist","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/AcKRB8wDpdaN6v6ru/interpreting-gpt-the-logit-lens",0,"","interpretability"],["Safe Scrambling?","Hoagy","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jHzb5SmviScXdtT2m/safe-scrambling",0,"",""],["Updates and additions to \"Embedded Agency\"","Rob Bensinger and abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9vYg8MyLL4cMMaPQJ/updates-and-additions-to-embedded-agency",0,"","theory"],["A Framework for Improving Scholarly Neural Network Diagrams","Guy Clarke Marshall and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.12566",0,"","evals"],["Basic Inframeasure Theory","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YAa4qcMyoucRS2Ykr/basic-inframeasure-theory",0,"","theory"],["Belief Functions And Decision Theory","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/e8qFDMzs2u9xf5ie6/belief-functions-and-decision-theory",0,"","theory"],["Model splintering: moving from one imperfect model to another","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/k54rgSg7GcjtXnMHX/model-splintering-moving-from-one-imperfect-model-to-another-1",0,"","scalable-oversight deception"],["Preface to the sequence on economic growth","Matthew Barnett","2020","blog","LessWrong","www.lesswrong.com/posts/2ADWcxNjQN3pbywtg/preface-to-the-sequence-on-economic-growth",0,"","forecasting"],["Proofs Section 1.1 (Initial results to LF-duality)","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PTcktJADsAmpYEjoP/proofs-section-1-1-initial-results-to-lf-duality",0,"",""],["Proofs Section 1.2 (Mixtures, Updates, Pushforwards)","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/b9jubzqz866CModHB/proofs-section-1-2-mixtures-updates-pushforwards",0,"",""],["Proofs Section 2.1 (Theorem 1, Lemmas)","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/xQYF3LR64NYn8vkoy/proofs-section-2-1-theorem-1-lemmas",0,"","theory"],["Proofs Section 2.2 (Isomorphism to Expectations)","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8tLPYYQJM8SwL2xn9/proofs-section-2-2-isomorphism-to-expectations",0,"","theory"],["Proofs Section 2.3 (Updates, Decision Theory)","Diffractor","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9ekP8FojvLa8Pr6P7/proofs-section-2-3-updates-decision-theory",0,"","theory"],["Self-classifying MNIST Digits","Ettore Randazzo and 4 others","2020","report","Distill","distill.pub/2020/selforg/mnist",0,"",""],["Technical model refinement formalism","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/89qWCy6yi2eeFGsRu/technical-model-refinement-formalism",0,"",""],["Thread: Differentiable Self-organizing Systems","Alexander Mordvintsev and 21 others","2020","report","Distill","distill.pub/2020/selforg",0,"",""],["[AN #114]: Theory-inspired safety solutions for powerful Bayesian RL agents","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/kxPiL4zNSPR249wsC/an-114-theory-inspired-safety-solutions-for-powerful",0,"","agents"],["Asya Bergal: Reasons you might think human-level AI is unlikely to happen soon","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/kJzPDbgmA8nrLqTgH/asya-bergal-reasons-you-might-think-human-level-ai-is",0,"","forecasting"],["Introduction To The Infra-Bayesianism Sequence","Diffractor and Vanessa Kosoy","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zB4f7QqKhBHa5b37a/introduction-to-the-infra-bayesianism-sequence",0,"","theory"],["nostalgebraist: Recursive Goodhart's Law","Kaj_Sotala","2020","blog","LessWrong","www.lesswrong.com/posts/PAB2ymaaqZ3iXYfYA/nostalgebraist-recursive-goodhart-s-law",0,"","goodharts-law"],["Singapore’s Technical AI Alignment Research Career Guide","Yi-Yang","2020","blog","EA Forum","forum.effectivealtruism.org/posts/fLroJGMbszAjYBSdE/singapore-s-technical-ai-alignment-research-career-guide-1",0,"",""],["What is the interpretation of the do() operator?","Bunthut","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/uwfstudGoNFSEtMAT/what-is-the-interpretation-of-the-do-operator",0,"",""],["Learning human preferences: black-box, white-box, and structured white-box access","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9rjW9rhyhJijHTM92/learning-human-preferences-black-box-white-box-and",0,"",""],["Forecasting Thread: AI Timelines","Amandango and 2 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hQysqfSEzciRazx8k/forecasting-thread-ai-timelines",0,"","forecasting"],["A Composable Specification Language for Reinforcement Learning Tasks","Kishor Jothimurugan and 2 others","2020","paper","In Advances in Neural Information Processing Systems, pp.\n  13041-13051. 2019","arxiv.org/abs/2008.09293",0,"",""],["Thoughts on the Feasibility of Prosaic AGI Alignment?","anonymous","2020","blog","LessWrong","www.lesswrong.com/posts/cDGhjZM8nccWyScTn/thoughts-on-the-feasibility-of-prosaic-agi-alignment",0,"",""],["Understanding View Selection for Contrastive Learning","Yonglong Tian","2020","report","ai.googleblog.com","ai.googleblog.com/2020/08/understanding-view-selection-for.html",0,"",""],["Universality Unwrapped","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/farherQcqFQXqRcvv/universality-unwrapped",0,"","deception"],["What's a Decomposable Alignment Topic?","Logan Riggs","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4bAd9mFBLAFxR3MSk/what-s-a-decomposable-alignment-topic",0,"",""],["Animal Rights, The Singularity, and Astronomical Suffering","sapphire","2020","blog","EA Forum","forum.effectivealtruism.org/posts/dfgyNc4ShWZCQWJiu/animal-rights-the-singularity-and-astronomical-suffering",0,"",""],["[AN #113]: Checking the ethical intuitions of large language models","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/tDDDZ2nZdvyziwSvv/an-113-checking-the-ethical-intuitions-of-large-language",0,"",""],["AI safety as featherless bipeds *with broad flat nails*","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gWxMZisqE2j2kHCd2/ai-safety-as-featherless-bipeds-with-broad-flat-nails",0,"",""],["Alex Irpan: \"My AI Timelines Have Sped Up\"","Vaniver","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WFx8iPDS4WaaHyCtL/alex-irpan-my-ai-timelines-have-sped-up",0,"","forecasting"],["Looking for adversarial collaborators to test our Debate protocol","Beth Barnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/w7mS6syTderWihHPM/looking-for-adversarial-collaborators-to-test-our-debate",0,"","debate deception"],["Deploying Lifelong Open-Domain Dialogue Learning","Kurt Shuster and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.08076",0,"","evals"],["Learning human preferences: optimistic and pessimistic scenarios","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6XLyM22PBd9qDtin8/learning-human-preferences-optimistic-and-pessimistic",0,"",""],["Mesa-Search vs Mesa-Control","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WmBukJkEFM72Xr397/mesa-search-vs-mesa-control",0,"",""],["Radical Probabilism","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/xJyY5QkQvNJpZLJRo/radical-probabilism-1",0,"","agents theory"],["A way to beat superrational/EDT agents?","Abhimanyu Pallavi Sudhir","2020","blog","LessWrong","www.lesswrong.com/posts/aayFmJEF5PycJWuvW/a-way-to-beat-superrational-edt-agents",0,"","agents theory"],["Forward and inverse reinforcement learning sharing network weights and hyperparameters","Eiji Uchibe and Kenji Doya","2020","paper","Neural Networks, December 2021, Pages 138-153","arxiv.org/abs/2008.07284",0,"","evals policy"],["Runtime-Safety-Guided Policy Repair","Weichao Zhou and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.07667",0,"","assurance policy"],["Goal-Directedness: What Success Looks Like","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jP4cx3TCweDngSLS6/goal-directedness-what-success-looks-like",0,"",""],["Search versus design","Alex Flint","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/r3NHPD3dLFNk9QE2Y/search-versus-design-1",0,"","interpretability evals robustness"],["Adversarial Policies: Attacking Deep Reinforcement Learning.","Adam Gleave and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/1905.10615",0,"","agents policy"],["Conservative agency via attainable utility preservation..","Alexander Matt Turner and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/1902.09725",0,"",""],["Incomplete Contracting and AI Alignment.","Dylan Hadfield-Menell and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/1804.04268",0,"","agents"],["LESS is More: Rethinking Probabilistic Models of Human Behavior.","Andreea Bobu and 9 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.04465",0,"","sandbagging robustness"],["Mapping Out Alignment","Logan Riggs and 4 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jeiz7WfCnGQWoShkT/mapping-out-alignment",0,"",""],["My Understanding of Paul Christiano's Iterated Amplification AI Safety Research Agenda","Chi Nguyen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PT8vSxsusqWuN7JXp/my-understanding-of-paul-christiano-s-iterated-amplification",0,"","scalable-oversight"],["My Understanding of Paul Christiano's Iterated Amplification AI Safety Research Agenda","Chi","2020","blog","EA Forum","forum.effectivealtruism.org/posts/2ZeHrfJr9uHHJ2e8J/my-understanding-of-paul-christiano-s-iterated-amplification",0,"","scalable-oversight"],["On the Geometry of Adversarial Examples.","Marc Khoury and Dylan Hadfield-Menell","2020","paper","arXiv preprint","arxiv.org/abs/1811.00525",0,"","deception robustness"],["Self-Similarity Experiment","Dawn Drescher","2020","blog","LessWrong","www.lesswrong.com/posts/YNFtQx3nCcRPCPRZX/self-similarity-experiment",0,"","theory"],["SQIL: Imitation Learning via Regularized Behavioral Cloning..","Siddharth Reddy and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/1905.11108",0,"","policy"],["What are you optimizing for? Aligning Recommender Systems with Human Values.","Jonathan Stray and 2 others","2020","report","participatoryml.github.io","participatoryml.github.io/papers/2020/42.pdf",0,"","evals monitoring"],["A rational model of sequential self-assessment.","Rachel Jansen and 3 others","2020","report","cogsci.mindmodeling.org","cogsci.mindmodeling.org/2020/papers/0073/index.html",0,"",""],["A Rational Reinterpretation of Dual-Process Theories.","Smitha Milli and 2 others","2020","report","researchgate.net","www.researchgate.net/profile/Falk_Lieder/publication/323497836_A_Rational_Reinterpretation_of_Dual-Process_Theories/links/5c3a420492851c22a370c92e/A-Rational-Reinterpretation-of-Dual-Process-Theories.pdf",0,"",""],["Adaptive Autonomous Secure Cyber Systems.","Sushil Jajodia and 7 others","2020","report","link.springer.com","link.springer.com/book/10.1007%2F978-3-030-33432-1",0,"",""],["Advancing rational analysis to the algorithmic level.","Falk Lieder and 2 others","2020","report","cambridge.org","www.cambridge.org/core/journals/behavioral-and-brain-sciences/article/advancing-rational-analysis-to-the-algorithmic-level/F608CB0ABB8651A920C4BFBCA3E48945",0,"",""],["AI Research Considerations for Human Existential Safety (ARCHES).","Andrew Critch and David Krueger","2020","report","acritch.com","acritch.com/media/arches.pdf",0,"",""],["Aligning AI With Shared Human Values.","Dan Hendrycks and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.02275",0,"","agents"],["Aligning with Heterogeneous Preferences for Kidney Exchange.","Rachel Freedman","2020","paper","arXiv preprint","arxiv.org/abs/2006.09519",0,"","evals"],["Artificial Intelligence: A Modern Approach (Textbook, 4th Edition).","Stuart Russell","2020","report","pearson.com","www.pearson.com/us/higher-education/program/Russell-Artificial-Intelligence-A-Modern-Approach-4th-Edition/PGM1263338.html",0,"",""],["Assessing Mathematics Misunderstandings via Bayesian Inverse Planning.","Anna N and 4 others","2020","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/10.1111/cogs.12900",0,"",""],["AugMix: A Simple Data Processing Method to Improve Robustness and Uncertainty.","Dan Hendrycks and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/1912.02781",0,"","robustness"],["Choice Set Misspecification in Reward Inference.","Rachel Freedman and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2101.07691",0,"","rlhf"],["Combining experts’ causal judgments.","Dalal Alrajeh and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.10180",0,"","policy"],["Data-Driven, Photorealistic Social Face-Trait Encoding, Prediction, and Manipulation Using Deep Neural Networks.","Alexander Todorov and 4 others","2020","report","collaborate.princeton.edu","collaborate.princeton.edu/en/publications/data-driven-photorealistic-social-face-trait-encoding-prediction-",0,"",""],["Decentralized Reinforcement Learning: Global Decision-Making via Local Economic Transactions.","Michael Chang and 5 others","2020","report","proceedings.mlr.press","proceedings.mlr.press/v119/chang20b.html",0,"",""],["DERAIL: Diagnostic Environments for Reward And Imitation Learning.","Pedro Freire and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.01365",0,"","benchmarks agents"],["Downloading Culture.zip: Social learning by program induction.","Max Kleiman-Weiner and 6 others","2020","report","cogsci.mindmodeling.org","cogsci.mindmodeling.org/2020/papers/0365/index.html",0,"",""],["Dynamic Awareness.","Joseph Y Halpern and Evan Piermont","2020","paper","arXiv preprint","arxiv.org/abs/2007.02823",0,"","agents"],["Efficient Iterative Linear-Quadratic Approximations for Nonlinear Multi-Player General-Sum Differential Games.","David Fridovich-Keil and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/1909.04694",0,"","benchmarks deception agents"],["Emergent Complexity and Zero-shot Transfer via Unsupervised Environment Design.","Michael Dennis and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.02096",0,"","agents"],["Extracting low-dimensional psychological representations from convolutional neural networks.","Aditi Jha and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.14363",0,"","deception"],["Generalizing meanings from partners to populations: Hierarchical inference supports convention formation on networks.","Robert D and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.01510",0,"","evals agents robustness"],["Hidden Community Detection on Two-layer Stochastic Models: a Theoretical Perspective.","Jialu Bao and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.05919",0,"",""],["How Should an Agent Practice?.","Janarthanan Rajendran and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/1912.07045",0,"","evals agents"],["Interpretable and Pedagogical Examples.","Smitha Milli and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/1711.00694",0,"","interpretability evals"],["Measuring Massive Multitask Language Understanding.","Dan Hendrycks and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2009.03300",0,"","evals benchmarks policy"],["misc raw responses to a tract of Critical Rationalism","mako yass","2020","blog","LessWrong","www.lesswrong.com/posts/bsteawFidASBXiywa/misc-raw-responses-to-a-tract-of-critical-rationalism",0,"","governance"],["Multi-Principal Assistance Games.","Arnaud Fickinger and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.09540",0,"","agents"],["Patient-adaptable intracranial pressure morphology analysis using a probabilistic model-based approach.","Paria Rashidinejad and 2 others","2020","report","pubmed.ncbi.nlm.nih.gov","pubmed.ncbi.nlm.nih.gov/32992304/",0,"",""],["People Do Not Just Plan,They Plan to Plan.","Mark K and 8 others","2020","report","ojs.aaai.org","ojs.aaai.org//index.php/AAAI/article/view/5485",0,"",""],["Pretrained Transformers Improve Out-of-Distribution Robustness.","Dan Hendrycks and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.06100",0,"","robustness"],["Reconciling novelty and complexity through a rational analysis of curiosity.","R and 4 others","2020","report","psycnet.apa.org","psycnet.apa.org/record/2019-79765-001",0,"",""],["Resource-rational Task Decomposition to Minimize Planning Costs.","Carlos G and 5 others","2020","report","cogsci.mindmodeling.org","cogsci.mindmodeling.org/2020/papers/0746/index.html",0,"",""],["Scaling up psychology via Scientific Regret Minimization.","Mayank Agrawal and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/1910.07581",0,"","deception agents"],["SLIP: Learning to predict in unknown dynamical systems with long-term memory.","Paria Rashidinejad and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2010.05899",0,"","forecasting"],["Solving hard AI planning instances using curriculum-driven deep reinforcement learning.","Dieqiao Feng and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.02689",0,"",""],["Sparse Graphical Memory for Robust Planning.","Scott Emmons and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.06417",0,"","agents"],["Structure Learning for Approximate Solution of Many-Player Games.","Zun Li and 2 others","2020","report","aaai.org","aaai.org/ojs/index.php/AAAI/article/view/5586",0,"",""],["The Efficiency of Human Cognition Reflects Planned Information Processing.","Mark K and 8 others","2020","report","aaai.org","www.aaai.org/Papers/AAAI/2020GB/AAAI-HoM.5623.pdf",0,"",""],["The MAGICAL Benchmark for Robust Imitation.","Sam Toyer and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2011.00401",0,"","evals benchmarks robustness"],["The Many Faces of Robustness: A Critical Analysis of Out-of-Distribution Generalization.","Dan Hendrycks and 12 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.16241",0,"","evals benchmarks robustness"],["The method of loci is an optimal policy for memory search.","Qiong Zhang and 3 others","2020","report","cogsci.mindmodeling.org","cogsci.mindmodeling.org/2020/papers/0243/index.html",0,"","policy"],["Translucent players: Explaining cooperative behavior in social dilemmas.","Valerio Capraro and Joseph Y Halpern","2020","report","journals.sagepub.com","journals.sagepub.com/doi/abs/10.1177/1043463119885102",0,"",""],["Understanding Learned Reward Functions.","Eric J and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.05862",0,"","interpretability agents"],["Value-laden Disciplinary Shifts in Machine Learning.","Ravit Dotan and Smitha Milli","2020","paper","arXiv preprint","arxiv.org/abs/1912.01172",0,"","evals"],["[AN #112]: Engineering a Safer World","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bP6KA2JJQMke8H4Au/an-112-engineering-a-safer-world",0,"",""],["August 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/08/13/august-2020-newsletter/",0,"",""],["Considerations, Good Practices, Risks and Pitfalls in Developing AI Solutions Against COVID-19","Alexandra Luccioni and 4 others","2020","paper","Harvard CRCS Workshop on AI for Social Good, United States, 2020","arxiv.org/abs/2008.09043",0,"","robustness"],["Alignment By Default","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Nwgdq6kHke5LY692J/alignment-by-default",0,"","scalable-oversight"],["Blog post: A tale of two research communities","Aryeh Englander","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/FzF4Xok63ZCZNjmGY/blog-post-a-tale-of-two-research-communities",0,"",""],["Matt Botvinick on the spontaneous emergence of learning algorithms","Adam Scholl","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Wnqua6eQkewL3bqsF/matt-botvinick-on-the-spontaneous-emergence-of-learning",0,"",""],["Strong implication of preference uncertainty","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/q9GZyfm8xKAD2BGdi/strong-implication-of-preference-uncertainty",0,"",""],["Book review: Architects of Intelligence by Martin Ford (2018)","Ofer","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/iZS3am4acMh8g4Ycb/book-review-architects-of-intelligence-by-martin-ford-2018",0,"","governance"],["Will OpenAI's work unintentionally increase existential risks related to AI?","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/CD8gcugDu5z2Eeq7k/will-openai-s-work-unintentionally-increase-existential",0,"",""],["Adapting a kidney exchange algorithm to align with human values.","Rachel Freedman and 5 others","2020","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0004370220300229",0,"",""],["Approximate Causal Abstractions.","Sander Beckers and 2 others","2020","report","proceedings.mlr.press","proceedings.mlr.press/v115/beckers20a/beckers20a.pdf",0,"",""],["ASNets: Deep Learning for Generalised Planning.","Sam Toyer and 3 others","2020","report","jair.org","www.jair.org/index.php/jair/article/view/11633",0,"",""],["AvE: Assistance via Empowerment.","Yuqing Du and 6 others","2020","report","proceedings.neurips.cc","proceedings.neurips.cc/paper/2020/hash/30de9ece7cf3790c8c39ccff1a044209-Abstract.html",0,"",""],["Bounded Rationality in Las Vegas: Probabilistic Finite Automata Play Multi-Armed Bandits.","Xinming Liu and Joseph Halpern","2020","report","proceedings.mlr.press","proceedings.mlr.press/v124/liu20c/liu20c.pdf",0,"",""],["Cognitive prostheses for goal achievement.","Falk Lieder and 6 others","2020","report","nature.com","www.nature.com/articles/s41562-019-0672-9",0,"",""],["Exploring AI Safety in Degrees: Generality, Capability and Control","John Burden and Jose Hernandez-Orallo","2020","report","ceur-ws.org","ceur-ws.org/Vol-2560/paper21.pdf",0,"",""],["Forecasting AI Progress: A Research Agenda","rossg and axioman","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qsay72ct2KTmJ2hxc/forecasting-ai-progress-a-research-agenda",0,"","forecasting"],["How to Be Helpful to Multiple People at Once.","Vael Gates and 4 others","2020","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/Gates2020.pdf",0,"",""],["Inconsistency evaluation in pairwise comparison using norm-based distances.","Michele Fedrizzi and 2 others","2020","report","link.springer.com","link.springer.com/article/10.1007/s10203-020-00304-9",0,"","evals"],["Learning Rewards from Linguistic Feedback.","Theodore R and 8 others","2020","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/sumerslearning.pdf",0,"",""],["Market Manipulation: An Adversarial Learning Framework for Detection and Evasion.","Xintong Wang and Michael P Wellman","2020","report","www-personal.umich.edu","www-personal.umich.edu/~xintongw/papers/advgan2020ijcai.pdf",0,"",""],["Predicting responsibility judgments from dispositional inferences and causal attributions.","Antonia Langenhoff and 6 others","2020","report","psyarxiv.com","psyarxiv.com/63zvw",0,"",""],["Preference learning along multiple criteria: A game-theoretic perspective.","Kush Bhatia and 7 others","2020","report","proceedings.neurips.cc","proceedings.neurips.cc/paper/2020/hash/52f4691a4de70b3c441bca6c546979d9-Abstract.html",0,"",""],["Rational use of episodic and working memory: A normative account of prospective memory.","Ida Momennejad and 5 others","2020","report","biorxiv.org","www.biorxiv.org/content/biorxiv/early/2019/03/17/580324.full.pdf",0,"",""],["Sparse Skill Coding: Learning Behavioral Hierarchies with Sparse Codes.","Sophia Sanborn and 3 others","2020","report","openreview.net","openreview.net/forum?id=Hygv3xrtDr",0,"",""],["Special issue on autonomous agents modelling other agents: Guest editorial.","Stefano V and 4 others","2020","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0004370220300515",0,"","agents"],["SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards..","Siddharth Reddy and 3 others","2020","report","openreview.net","openreview.net/forum?id=S1xKd24twB",0,"",""],["What Can Learned Intrinsic Rewards Capture?.","Zeyu Zheng and 7 others","2020","report","proceedings.mlr.press","proceedings.mlr.press/v119/zheng20b/zheng20b.pdf",0,"",""],["What the Baldwin Effect affects depends on the nature of plasticity.","Thomas J and 6 others","2020","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0010027719303397",0,"",""],["10/50/90% chance of GPT-N Transformative AI?","human_generated_text","2020","blog","LessWrong","www.lesswrong.com/posts/z8DRKBKvM9JXrqbWH/10-50-90-chance-of-gpt-n-transformative-ai",0,"","forecasting"],["Non-Adversarial Imitation Learning and its Connections to Adversarial Methods","Oleg Arenz and Gerhard Neumann","2020","paper","arXiv preprint","arxiv.org/abs/2008.03525",0,"","agents policy"],["The Fusion Power Generator Scenario","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2NaAhMPGub8F2Pbr7/the-fusion-power-generator-scenario",0,"",""],["Towards a Formalisation of Logical Counterfactuals","Bunthut","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/A9d8npg59GRd83c5k/towards-a-formalisation-of-logical-counterfactuals",0,"",""],["Impact of meta-roles on the evolution of organisational institutions","Amir Hosein Afshar Sedigh and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.04096",0,"","agents"],["Analyzing the Problem GPT-3 is Trying to Solve","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XdnCyorFzYskS7EtP/analyzing-the-problem-gpt-3-is-trying-to-solve",0,"",""],["Decoupling Exploration and Exploitation for Meta-Reinforcement Learning without Sacrifices","Evan Zheran Liu and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.02790",0,"","agents robustness"],["The Whiteness of AI","Stephen Cave and Kanta Dihal","2020","report","doi.org","doi.org/10.1007/s13347-020-00415-6",0,"",""],["[AN #111]: The Circuits hypotheses for deep learning","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/5CApLZiHGkt37nRQ2/an-111-the-circuits-hypotheses-for-deep-learning",0,"","mechanistic-interpretability"],["Measuring hardware overhang","hippke","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/75dnjiD8kv2khe9eQ/measuring-hardware-overhang",0,"","forecasting"],["Collecting the Public Perception of AI and Robot Rights","Gabriel Lima and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.01339",0,"","agents robustness"],["Forecasting AI Progress: A Research Agenda","Ross Gruetzemacher and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.01848",0,"","forecasting"],["Infinite Data/Compute Arguments in Alignment","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/7CJBiHYxebTmMfGs3/infinite-data-compute-arguments-in-alignment",0,"",""],["Interpretability in ML: A Broad Overview","anonymous","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/57fTWCpsAyjeAimTp/interpretability-in-ml-a-broad-overview-2",0,"","interpretability"],["(Ir)rationality of Pascal's wager","filozof3377@gmial.com","2020","blog","LessWrong","www.lesswrong.com/posts/KbEbgbXL64Rqhsi9k/ir-rationality-of-pascal-s-wager",0,"","theory"],["AI Risk: Increasing Persuasion Power","kewlcats","2020","blog","EA Forum","forum.effectivealtruism.org/posts/prpDSEQXgffZtvPST/ai-risk-increasing-persuasion-power",0,"",""],["Is GPT-3 the death of the paperclip maximizer?","matthias_samwald","2020","blog","EA Forum","forum.effectivealtruism.org/posts/gYCjGx6fnJSSMva4v/is-gpt-3-the-death-of-the-paperclip-maximizer",0,"",""],["Three mental images from thinking about AGI debate & corrigibility","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WjY9y7r52vaNZ2WmH/three-mental-images-from-thinking-about-agi-debate-and",0,"",""],["What are the most important papers/post/resources to read to understand more of GPT-3?","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/m23A54nFL5yeDRukL/what-are-the-most-important-papers-post-resources-to-read-to",0,"",""],["What do we do if AI doesn't take over the world, but still causes a significant global problem?","James_Banks","2020","blog","EA Forum","forum.effectivealtruism.org/posts/AjxZ8RTNPZTmwTh2j/what-do-we-do-if-ai-doesn-t-take-over-the-world-but-still",0,"","governance"],["Inner Alignment: Explain like I'm 12 Edition","Rafael Harth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/AHhCrJ2KpTjsCSwbt/inner-alignment-explain-like-i-m-12-edition",0,"",""],["Power as Easily Exploitable Opportunities","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eqov4SEYEbeFMXegR/power-as-easily-exploitable-opportunities",0,"","instrumental-convergence"],["Testing the Automation Revolution Hypothesis","Keller Scholl and Robin Hanson","2020","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0165176520301919",0,"",""],["\"Go west, young man!\" - Preferences in (imperfect) maps","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pfmFe5fgEn2weJuer/go-west-young-man-preferences-in-imperfect-maps",0,"",""],["On Single Point Forecasts for Fat-Tailed Variables","Nassim Nicholas Taleb12 and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.16096",0,"","forecasting"],["Is the work on AI alignment relevant to GPT?","Richard_Kennaway","2020","blog","LessWrong","www.lesswrong.com/posts/dPcKrfEi87Zzr7w6H/is-the-work-on-ai-alignment-relevant-to-gpt",0,"",""],["The academic contribution to AI safety seems large","Gavin","2020","blog","EA Forum","forum.effectivealtruism.org/posts/8ErtxW7FRPGMtDqJy/the-academic-contribution-to-ai-safety-seems-large",0,"",""],["What if memes are common in highly capable minds?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6iedrXht3GpKTQWRF/what-if-memes-are-common-in-highly-capable-minds",0,"","agents"],["[AN #110]: Learning features from human feedback to enable reward learning","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/P6eWEMCrSjbuWwESk/an-110-learning-features-from-human-feedback-to-enable",0,"","rlhf"],["Engaging Seriously with Short Timelines","sapphire","2020","blog","LessWrong","www.lesswrong.com/posts/dZoXpSa3WehwqCf2m/engaging-seriously-with-short-timelines",0,"","forecasting"],["Learning the prior and generalization","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YhQr36yGkhe6x8Fyn/learning-the-prior-and-generalization",0,"",""],["Rohin Shah: What’s been happening in AI alignment?","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/nqTdRNngCGDD54owu/rohin-shah-what-s-been-happening-in-ai-alignment",0,"",""],["The \"best predictor is malicious optimiser\" problem","Donald Hobson","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ARGXciEGtuhMKtYb8/the-best-predictor-is-malicious-optimiser-problem",0,"",""],["What Failure Looks Like: Distilling the Discussion","Ben Pace","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6jkGf5WEKMpMFXZp2/what-failure-looks-like-distilling-the-discussion",0,"",""],["Does the lottery ticket hypothesis suggest the scaling hypothesis?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/wFJqi75y9eW8mf8TR/does-the-lottery-ticket-hypothesis-suggest-the-scaling",0,"",""],["FHI Report: How Will National Security Considerations Affect Antitrust Decisions in AI? An Examination of Historical Precedents","Cullen","2020","blog","EA Forum","forum.effectivealtruism.org/posts/qotAq6NeabqvEsXz2/fhi-report-how-will-national-security-considerations-affect",0,"","governance policy"],["Probability that other architectures will scale as well as Transformers?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/kpK6854ArgwySuv7D/probability-that-other-architectures-will-scale-as-well-as",0,"","forecasting"],["To what extent are the scaling properties of Transformer networks exceptional?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/KyM9p6q5SELM3Ncdu/to-what-extent-are-the-scaling-properties-of-transformer-1",0,"",""],["What happens to variance as neural network training is scaled? What does it imply about \"lottery tickets\"?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/kFm9ZMreqeNYpg8m8/what-happens-to-variance-as-neural-network-training-is",0,"","forecasting"],["What specific dangers arise when asking GPT-N to write an Alignment Forum post?","Matthew Barnett","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Et2pWrj4nWfdNAawh/what-specific-dangers-arise-when-asking-gpt-n-to-write-an",0,"",""],["AI and Efficiency","DragonGod","2020","blog","LessWrong","www.lesswrong.com/posts/AmaoxqZTJZnWgmtCj/ai-and-efficiency",0,"","forecasting"],["Are we in an AI overhang?","Andy Jones","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/N6vZEnCn6A95Xn39p/are-we-in-an-ai-overhang",0,"","forecasting"],["Combining Deep Reinforcement Learning and Search for Imperfect-Information Games","Noam Brown and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.13544",0,"","deception policy"],["Generalizing the Power-Seeking Theorems","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/nyDnLif4cjeRe9DSv/generalizing-the-power-seeking-theorems",0,"","instrumental-convergence power-seeking agents"],["Developmental Stages of GPTs","orthonormal","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3nDR23ksSQJ98WNDm/developmental-stages-of-gpts",0,"","forecasting"],["Automated Database Indexing using Model-free Reinforcement Learning","Gabriel Paludo Licks and Felipe Meneguzzi","2020","paper","arXiv preprint","arxiv.org/abs/2007.14244",0,"","evals"],["Constraints from naturalized ethics.","Charlie Steiner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/xN8MTRN7GchFJB8WJ/constraints-from-naturalized-ethics",0,"",""],["Markus Anderljung and Ben Garfinkel: Fireside chat on AI governance","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/aeJB4qAWBxcvtZHad/markus-anderljung-and-ben-garfinkel-fireside-chat-on-ai",0,"","governance"],["Bridging the Imitation Gap by Adaptive Insubordination","Luca Weihs and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.12173",0,"","agents"],["Can you get AGI from a Transformer?","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SkcM4hwgH3AP6iqjs/can-you-get-agi-from-a-transformer",0,"","forecasting"],["Improving Competence for Reliable Autonomy","Connor Basich and 4 others","2020","paper","EPTCS 319, 2020, pp. 37-53","arxiv.org/abs/2007.11740",0,"","agents"],["Optimizing arbitrary expressions with a linear number of queries to a Logical Induction Oracle (Cartoon Guide)","Donald Hobson","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/H32NbFcqjTxy2pvaq/optimizing-arbitrary-expressions-with-a-linear-number-of",0,"","theory"],["Scrutinizing AI Risk (80K, #81) - v. quick summary","Ben","2020","blog","EA Forum","forum.effectivealtruism.org/posts/hBP67ZkaBPNrJSpWT/scrutinizing-ai-risk-80k-81-v-quick-summary",0,"",""],["Toward Campus Mail Delivery Using BDI","Chidiebere Onyedinma and 2 others","2020","paper","EPTCS 319, 2020, pp. 127-143","arxiv.org/abs/2007.16089",0,"","agents"],["Why the Orthogonality Thesis's veracity is not the point:","Antoine de Scorraille","2020","blog","EA Forum","forum.effectivealtruism.org/posts/RRaN57QAw8XNi9RXN/why-the-orthogonality-thesis-s-veracity-is-not-the-point",0,"",""],["[AN #109]: Teaching neural nets to generalize the way humans would","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TWdnCi4kPjTapYjh6/an-109-teaching-neural-nets-to-generalize-the-way-humans",0,"",""],["Intellectual Diversity in AI Safety","KR","2020","blog","EA Forum","forum.effectivealtruism.org/posts/jAWSicEi3PD8JHmac/intellectual-diversity-in-ai-safety",0,"",""],["Weak HCH accesses EXP","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/CtGH3yEoo4mY2taxe/weak-hch-accesses-exp",0,"","scalable-oversight"],["$1000 bounty for OpenAI to show whether GPT3 was \"deliberately\" pretending to be stupider than it is","jacobjacob","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/H9knnv8BWGKj6dZim/usd1000-bounty-for-openai-to-show-whether-gpt3-was",0,"",""],["[Preprint] The Computational Limits of Deep Learning","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/buhaT2pxsfLrknzxT/preprint-the-computational-limits-of-deep-learning",0,"",""],["AI Benefits Post 5: Outstanding Questions on Governing Benefits","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8e3FmHY4598SJ9PNL/ai-benefits-post-5-outstanding-questions-on-governing",0,"",""],["Alignment As A Bottleneck To Usefulness Of GPT-3","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BnDF5kejzQLqd5cjH/alignment-as-a-bottleneck-to-usefulness-of-gpt-3",0,"",""],["Competition: Amplify Rohin’s Prediction on AGI researchers & Safety Concerns","stuhlmueller","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Azqmzp5JoXJihMcr4/competition-amplify-rohin-s-prediction-on-agi-researchers",0,"","forecasting"],["How strong is the evidence of unaligned AI systems causing harm?","BrownHairedEevee","2020","blog","EA Forum","forum.effectivealtruism.org/posts/e9Q94Mq6LTjSAgujY/how-strong-is-the-evidence-of-unaligned-ai-systems-causing",0,"",""],["Artificial Intelligence is stupid and causal reasoning won't fix it","John Mark Bishop","2020","paper","arXiv preprint","arxiv.org/abs/2008.07371",0,"","deception"],["Learning Values in Practice","Stuart_Armstrong","2020","blog","LessWrong","www.lesswrong.com/posts/Xy2AYxpWqJWedFfcD/learning-values-in-practice",0,"",""],["Parallels Between AI Safety by Debate and Evidence Law","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SrsH2MyyH8MqH9QSr/parallels-between-ai-safety-by-debate-and-evidence-law",0,"","debate"],["Parallels Between AI Safety by Debate and Evidence Law","Cullen O'Keefe","2020","report","cullenokeefe.com","cullenokeefe.com/blog/debate-evidence",0,"",""],["To what extent is GPT-3 capable of reasoning?","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/L5JSMZQvkBAx9MD5A/to-what-extent-is-gpt-3-capable-of-reasoning",0,"","forecasting"],["What Would I Do? Self-prediction in Simple Algorithms","Scott Garrabrant","2020","blog","LessWrong","www.lesswrong.com/posts/PiXS9kE4qX68KveCt/what-would-i-do-self-prediction-in-simple-algorithms",0,"","theory"],["Collection of GPT-3 results","Kaj_Sotala","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Hee7w2paEzHsD6mn/collection-of-gpt-3-results",0,"",""],["Cool linguistic purisms","Tamsin Leake","2020","blog","carado.moe","carado.moe/linguistic-purisms.html",0,"",""],["Modulation of viability signals for self-regulatory control","Alvaro Ovalle and Simon M. Lucas","2020","paper","arXiv preprint","arxiv.org/abs/2007.09297",0,"","evals agents"],["Why is pseudo-alignment \"worse\" than other ways ML can fail to generalize?","nostalgebraist","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TSmgTGaLyhL965jX6/why-is-pseudo-alignment-worse-than-other-ways-ml-can-fail-to",0,"",""],["Environments as a bottleneck in AGI development","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/vqpEC3MPioHX7bv4t/environments-as-a-bottleneck-in-agi-development",0,"","agents forecasting robustness"],["Progress/decline in fields","Tamsin Leake","2020","blog","carado.moe","carado.moe/progress-decline.html",0,"",""],["Technologies for Trustworthy Machine Learning: A Survey in a Socio-Technical Context","Ehsan Toreini and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.08911",0,"","interpretability policy"],["[AN #107]: The convergent instrumental subgoals of goal-directed agents","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pjTF49Rnc878jZSAZ/an-107-the-convergent-instrumental-subgoals-of-goal-directed",0,"","agents"],["[AN #108]: Why we should scrutinize arguments for AI risk","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/T5awG3XQKJtprABsy/an-108-why-we-should-scrutinize-arguments-for-ai-risk",0,"",""],["A list of good heuristics that the case for AI X-risk fails","Aaron Gertler","2020","blog","EA Forum","forum.effectivealtruism.org/posts/2RurEJXi5PqbEsCZb/a-list-of-good-heuristics-that-the-case-for-ai-x-risk-fails",0,"","robustness"],["Alignment proposals and complexity classes","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/N64THGX7XNCqRtvPG/alignment-proposals-and-complexity-classes",0,"","scalable-oversight debate"],["Artificial Interdisciplinarity: Artificial Intelligence for Research on Complex Societal Problems","Seth D. Baum","2020","report","link.springer.com","link.springer.com/10.1007/s13347-020-00416-5",0,"",""],["LogiQA: A Challenge Dataset for Machine Reading Comprehension with Logical Reasoning","Jian Liu and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.08124",0,"","benchmarks"],["Failures of Contingent Thinking","Evan Piermont and Peio Zuazo-Garin","2020","paper","arXiv preprint","arxiv.org/abs/2007.07703",0,"","benchmarks agents"],["How should AI debate be judged?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/m7oGxvouzzeQKiGJH/how-should-ai-debate-be-judged",0,"","debate"],["New paper: AGI Agent Safety by Iteratively Improving the Utility Function","Koen.Holtman","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/HWRR8YzuM63yZyTPG/new-paper-agi-agent-safety-by-iteratively-improving-the",0,"","agents"],["AI Benefits Post 4: Outstanding Questions on Selecting Benefits","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/upYKjwjC67ovWKKMo/ai-benefits-post-4-outstanding-questions-on-selecting",0,"",""],["The Goldbach conjecture is probably correct; so was Fermat's last theorem","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/iNFZG4d9W848zsgch/the-goldbach-conjecture-is-probably-correct-so-was-fermat-s",0,"",""],["What are the mostly likely ways AGI will emerge?","Craig Quiter","2020","blog","LessWrong","www.lesswrong.com/posts/mvnEbSScBHpwxoGLT/what-are-the-mostly-likely-ways-agi-will-emerge",0,"","forecasting"],["3-P Group optimal for discussion?","AiresJL","2020","blog","LessWrong","www.lesswrong.com/posts/KJQjXAsvNkKmRkiXm/3-p-group-optimal-for-discussion-1",0,"","theory"],["AMA or discuss my 80K podcast episode: Ben Garfinkel, FHI researcher","bgarfinkel","2020","blog","EA Forum","forum.effectivealtruism.org/posts/7gxtXrMeqw78ZZeY9/ama-or-discuss-my-80k-podcast-episode-ben-garfinkel-fhi",0,"",""],["Null-boxing Newcomb’s Problem","Yitz","2020","blog","LessWrong","www.lesswrong.com/posts/fbjNLjNd4zRbY9Wg2/null-boxing-newcomb-s-problem-2",0,"",""],["Does generality pay? GPT-3 can provide preliminary evidence.","BrownHairedEevee","2020","blog","EA Forum","forum.effectivealtruism.org/posts/Wci5rtwGTg9tLAETN/does-generality-pay-gpt-3-can-provide-preliminary-evidence",0,"","governance forecasting"],["What counts as defection?","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8LEPDY36jBYpijrSw/what-counts-as-defection",0,"",""],["Meta Programming GPT: A route to Superintelligence?","dmtea","2020","blog","LessWrong","www.lesswrong.com/posts/zZLe74DvypRAf7DEQ/meta-programming-gpt-a-route-to-superintelligence",0,"",""],["A space of proposals for building safe advanced AI","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/S9GxuAEeQomnLkeNt/a-space-of-proposals-for-building-safe-advanced-ai",0,"","deception agents"],["Machine Learning Explainability for External Stakeholders","Umang Bhatt and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.05408",0,"","interpretability policy"],["Mesa-Optimizers vs “Steered Optimizers”","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SJXujr5a2NcoFebr4/mesa-optimizers-vs-steered-optimizers",0,"",""],["Talk: Key Issues In Near-Term AI Safety Research","Aryeh Englander","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/yijG7ptfqFBR8w885/talk-key-issues-in-near-term-ai-safety-research",0,"","evals assurance robustness"],["AI Research Considerations for Human Existential Safety (ARCHES)","habryka","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QmfjZMr9HxLwHcDQB/ai-research-considerations-for-human-existential-safety",0,"",""],["Arguments against myopic training","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/GqxuDtZvfgL2bEQ5v/arguments-against-myopic-training",0,"","reward-hacking agents policy robustness"],["Why is the impact penalty time-inconsistent?","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YjqwTepi53MyM4omT/why-is-the-impact-penalty-time-inconsistent",0,"",""],["Decolonial AI: Decolonial Theory as Sociotechnical Foresight in Artificial Intelligence","Shakir Mohamed and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.04068",0,"",""],["How \"honest\" is GPT-3?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/c3RsLTcxrvH4rXpBL/how-honest-is-gpt-3",0,"",""],["July 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/07/08/july-2020-newsletter/",0,"",""],["Mahendra Prasad: Rational group decision-making","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/PZ76HmcbNREuoAfgG/mahendra-prasad-rational-group-decision-making",0,"","policy"],["Sunday July 12 — talks by Scott Garrabrant, Alexflint, alexei, Stuart_Armstrong","jacobjacob and Ben Pace","2020","blog","LessWrong","www.lesswrong.com/posts/jRGmNi6CQFQDBvkub/sunday-july-12-talks-by-scott-garrabrant-alexflint-alexei",0,"",""],["What does it mean to apply decision theory?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/wgdfBtLmByaKYovYe/what-does-it-mean-to-apply-decision-theory",0,"","theory"],["Antitrust-Compliant AI Industry Self-Regulation","Cullen O’Keefe","2020","report","cullenokeefe.com","cullenokeefe.com/s/Antitrust-Compliant-AI-Industry-Self-Regulation.pdf",0,"","governance"],["Dynamic inconsistency of the inaction and initial state baseline","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/w8QBmgQwb83vDMXoz/dynamic-inconsistency-of-the-inaction-and-initial-state",0,"","agents"],["How Will National Security Considerations Affect Antitrust Decisions in AI? An Examination of Historical Precedents","Cullen O’Keefe","2020","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/How-Will-National-Security-Considerations-Affect-Antitrust-Decisions-in-AI-Cullen-OKeefe.pdf",0,"",""],["How Will National Security Considerations Affect Antitrust Decisions in AI? An Examination of Historical Precedents","Cullen O’Keefe","2020","report","governance.ai","www.governance.ai/research-paper/how-will-national-security-considerations-affect-antitrust-decisions-in-ai-an-examination-of-historical-precedents",0,"","policy"],["Reducing long-term risks from malevolent actors","David Althaus and Tobias Baumann","2020","report","longtermrisk.org","longtermrisk.org/reducing-long-term-risks-from-malevolent-actors/",0,"",""],["Robust Learning with Frequency Domain Regularization","Weiyu Guo and Yidong Ouyang","2020","paper","arXiv preprint","arxiv.org/abs/2007.03244",0,"","robustness"],["AI Benefits Post 3: Direct and Indirect Approaches to AI Benefits","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/q3xFWK3qcR7JGTxsv/ai-benefits-post-3-direct-and-indirect-approaches-to-ai",0,"",""],["Better priors as a safety problem","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/roA83jDvq7F2epnHK/better-priors-as-a-safety-problem",0,"",""],["Better priors as a safety problem","Paul Christiano","2020","report","ai-alignment.com","ai-alignment.com/better-priors-as-a-safety-problem-24aa1c300710",0,"",""],["Decentralized Reinforcement Learning: Global Decision-Making via Local Economic Transactions","Michael Chang and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.02382",0,"","agents"],["Learning the prior","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/SL9mKhgdmDKXmxwE4/learning-the-prior",0,"",""],["Learning the prior","Paul Christiano","2020","report","ai-alignment.com","ai-alignment.com/learning-the-prior-48f61b445c04",0,"",""],["Tradeoff between desirable properties for baseline choices in impact measures","Victoria Krakovna","2020","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2020/07/05/tradeoff-between-desirable-properties-for-baseline-choices-in-impact-measures/",0,"",""],["Customized Handling of Unintended Interface Operation in Assistive Robots","Deepak Gopinath and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.02092",0,"","evals"],["Tradeoff between desirable properties for baseline choices in impact measures","Vika","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/nLhfRpDutEdgr6PKe/tradeoff-between-desirable-properties-for-baseline-choices",0,"","agents policy"],["AI Unsafety via Non-Zero-Sum Debate","VojtaKovarik","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BRiMQELD5WYyvncTE/ai-unsafety-via-non-zero-sum-debate",0,"","debate agents"],["Research ideas to study humans with AI Safety in mind","Riccardo Volpato","2020","blog","LessWrong","www.lesswrong.com/posts/nqTkfrnE4CkbMtmHE/research-ideas-to-study-humans-with-ai-safety-in-mind",0,"",""],["Splitting Debate up into Two Subsystems","Nandi","2020","blog","LessWrong","www.lesswrong.com/posts/8v5kc4dKdeTvvEkc8/splitting-debate-up-into-two-subsystems",0,"",""],["Goals and short descriptions","Michele Campolo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/d4NgfKY3cq9yiBLSM/goals-and-short-descriptions",0,"",""],["The \"AI Debate\" Debate","michaelcohen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/L3QDs6of4Rb2TgpRD/the-ai-debate-debate",0,"",""],["Verifiably Safe Exploration for End-to-End Reinforcement Learning","Nathan Hunt and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.01223",0,"","policy"],["[AN #106]: Evaluating generalization ability of learned reward models","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dEqjwwvYtg9NEmZoq/an-106-evaluating-generalization-ability-of-learned-reward",0,"","evals"],["Artificial intelligence in a crisis needs ethics with urgency","Asaf Tzachor and 3 others","2020","report","nature.com","www.nature.com/articles/s42256-020-0195-0",0,"",""],["CHAI Newsletter #2 2020","CHAI","2020","report","drive.google.com","drive.google.com/file/d/10S9GD7IPaauOE4kBHGRnZxsrAkk8Q_S3/view?usp=sharing",0,"",""],["Evan Hubinger on Inner Alignment, Outer Alignment, and Proposals for Building Safe Advanced AI","Palus Astra","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qZGoHkRgANQpGHWnu/evan-hubinger-on-inner-alignment-outer-alignment-and",0,"",""],["FLI AI Alignment podcast: Evan Hubinger on Inner Alignment, Outer Alignment, and Proposals for Building Safe Advanced AI","evhub","2020","blog","EA Forum","forum.effectivealtruism.org/posts/LFN2NtfcKaCFayLtC/fli-ai-alignment-podcast-evan-hubinger-on-inner-alignment",0,"",""],["Unifying Model Explainability and Robustness via Machine-Checkable Concepts","Vedant Nanda and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2007.00251",0,"","robustness"],["Comparing AI Alignment Approaches to Minimize False Positive Risk","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eXNy48LxxfgETdtYB/comparing-ai-alignment-approaches-to-minimize-false-positive",0,"",""],["GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","Dmitry Lepikhin and 8 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.16668",0,"","benchmarks training-data"],["Web AI discussion Groups","Donald Hobson","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/omj76gXR67jsG4hxs/web-ai-discussion-groups",0,"",""],["AI Benefits Post 2: How AI Benefits Differs from AI Alignment & AI for Good","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Z5XXdQDxhpgiXASQW/ai-benefits-post-2-how-ai-benefits-differs-from-ai-alignment",0,"","robustness"],["How do takeoff speeds affect the probability of bad outcomes from AGI?","KR","2020","blog","LessWrong","www.lesswrong.com/posts/3cmbR4oeimCTJ67G3/how-do-takeoff-speeds-affect-the-probability-of-bad-outcomes",0,"","forecasting"],["Gary Marcus vs Cortical Uniformity","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8F8dagB4q4BzR5JNz/gary-marcus-vs-cortical-uniformity",0,"",""],["Have general decomposers been formalized?","Quinn","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xwu2PuLPWtJz8sLLA/have-general-decomposers-been-formalized",0,"",""],["Song Pairs that can be listened to together","Tamsin Leake","2020","blog","carado.moe","carado.moe/song-pairs.html",0,"",""],["AI safety via market making","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YWwzccGbcHMJMpT45/ai-safety-via-market-making",0,"","debate deception"],["AvE: Assistance via Empowerment","Yuqing Du and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.14796",0,"",""],["Does the Whole Exceed its Parts? The Effect of AI Explanations on Complementary Team Performance","Gagan Bansal and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.14779",0,"",""],["Is SGD a Bayesian sampler? Well, almost","Chris Mingard and 3 others","2020","paper","Journal of Machine Learning Research, 22 79 (2021), 1-64","arxiv.org/abs/2006.15191",0,"","robustness"],["Radical Probabilism [Transcript]","abramdemski and Ben Pace","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZM63n353vh2ag7z4p/radical-probabilism-transcript",0,"","theory"],["Some promising career ideas beyond 80,000 Hours' priority paths","Ardenlk","2020","blog","EA Forum","forum.effectivealtruism.org/posts/6x2MjPXhpPpnatJFQ/some-promising-career-ideas-beyond-80-000-hours-priority",0,"",""],["Widening the Pipeline in Human-Guided Reinforcement Learning with Explanation and Context-Aware Data Augmentation","Lin Guan and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.14804",0,"","evals agents"],["“Explaining” machine learning reveals policy challenges","Diane Coyle and Adrian Weller","2020","report","sciencemag.org","www.sciencemag.org/lookup/doi/10.1126/science.aba9647",0,"","policy"],["AI Governance Reading Group Guide","Alex HT","2020","blog","EA Forum","forum.effectivealtruism.org/posts/eLKX9bmra9ZR2AQzD/ai-governance-reading-group-guide",0,"","governance"],["Smooth Adversarial Training","Cihang Xie","2020","paper","arXiv preprint","arxiv.org/abs/2006.14536",0,"","robustness"],["[AN #105]: The economic trajectory of humanity, and what we might mean by optimization","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gWRJDwqHnmJhurXgo/an-105-the-economic-trajectory-of-humanity-and-what-we-might",0,"",""],["Abstraction, Evolution and Gears","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ahZQbxiPPpsTutDy2/abstraction-evolution-and-gears",0,"",""],["Compositional Explanations of Neurons","Jesse Mu and Jacob Andreas","2020","paper","arXiv preprint","arxiv.org/abs/2006.14032",0,"","interpretability deception robustness"],["Models, myths, dreams, and Cheshire cat grins","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hxzQoXjtLGRWPoLkE/models-myths-dreams-and-cheshire-cat-grins",0,"",""],["Quantifying Differences in Reward Functions","Adam Gleave and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.13900",0,"","evals policy"],["RL Unplugged: Benchmarks for Offline Reinforcement Learning","Caglar Gülçehre and 17 others","2020","blog","deepmind.com","www.deepmind.com/blog/rl-unplugged-benchmarks-for-offline-reinforcement-learning",0,"","benchmarks"],["The Dark Miracle of Optics","Suspended Reason","2020","blog","LessWrong","www.lesswrong.com/posts/zzt448rSfwdydinbZ/the-dark-miracle-of-optics",0,"","goodharts-law"],["Adversarial Soft Advantage Fitting: Imitation Learning without Policy Optimization","Paul Barde and 5 others","2020","paper","Advances in Neural Information Processing Systems 33 (2020)","arxiv.org/abs/2006.13258",0,"","evals policy"],["Modelling Continuous Progress","Sammy Martin","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/66FKFkWAugS8diydF/modelling-continuous-progress",0,"","forecasting"],["AI Benefits Post 1: Introducing “AI Benefits”","Cullen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/EGvtZMvSFELxoRqkZ/ai-benefits-post-1-introducing-ai-benefits",0,"",""],["Locality of goals","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/HkWB5KCJQ2aLsMzjt/locality-of-goals",0,"","reward-hacking power-seeking agents"],["Plausible cases for HRAD work, and locating the crux in the \"realism about rationality\" debate","riceissa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BGxTpdBGbwCWrGiCL/plausible-cases-for-hrad-work-and-locating-the-crux-in-the",0,"","agents"],["Safe Reinforcement Learning via Curriculum Induction","Matteo Turchetta and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.12136",0,"","agents policy"],["Stuart Russell on the flaws that make today’s AI architecture unsafe, and a new approach that could fix them","Robert Wiblin and 2 others","2020","report","80000hours.org","80000hours.org/podcast/episodes/stuart-russell-human-compatible-ai/",0,"",""],["The flaws that make today's AI architecture unsafe and a new approach that could fix it","80000_Hours","2020","blog","EA Forum","forum.effectivealtruism.org/posts/kN3HgzDajBRAyS3sS/the-flaws-that-make-today-s-ai-architecture-unsafe-and-a-new",0,"","governance"],["The Indexing Problem","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ABNjLr2H39g2oXqGb/the-indexing-problem",0,"",""],["Will AGI cause mass technological unemployment?","BrownHairedEevee","2020","blog","EA Forum","forum.effectivealtruism.org/posts/G9Zc3yaT2q2rZXBbL/will-agi-cause-mass-technological-unemployment",0,"","governance"],["Relevant pre-AGI possibilities","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zjhZpZi76kEBRnjiw/relevant-pre-agi-possibilities",0,"",""],["Relevant pre-AGI possibilities","kokotajlod","2020","blog","EA Forum","forum.effectivealtruism.org/posts/xGSw8gho7CJNXrPtf/relevant-pre-agi-possibilities",0,"","forecasting"],["The ground of optimization","Alex Flint","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/znfkdCoHMANwqc2WE/the-ground-of-optimization-1",0,"","agents robustness"],["Relevant pre-AGI possibilities","Daniel Kokotajlo","2020","blog","aiimpacts.org","aiimpacts.org/relevant-pre-agi-possibilities/",0,"","policy"],["[AN #104]: The perils of inaccessible information, and what we can learn about AI alignment from COVID","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/eE4QrWsdYQxNynbTM/an-104-the-perils-of-inaccessible-information-and-what-we",0,"",""],["IReEn: Reverse-Engineering of Black-Box Functions via Iterative Neural Program Synthesis","Hossein Hajipour and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.10720",0,"","interpretability agents"],["Big Self-Supervised Models are Strong Semi-Supervised Learners","Ting Chen and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.10029",0,"","robustness"],["Curve Detectors","Nick Cammarata and 5 others","2020","report","Distill","distill.pub/2020/circuits/curve-detectors",0,"",""],["Our take on CHAI’s research agenda in under 1500 words","Alex Flint","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qPoaA5ZSedivA4xJa/our-take-on-chai-s-research-agenda-in-under-1500-words",0,"",""],["Results of $1,000 Oracle contest!","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YbYFeZQWncy9Tzzq9/results-of-usd1-000-oracle-contest",0,"",""],["Unsupervised Learning of Visual Features by Contrasting Cluster Assignments","Mathilde Caron and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.09882",0,"",""],["Relating HCH and Logical Induction","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/R3HAvMGFNJGXstckQ/relating-hch-and-logical-induction",0,"","theory"],["What are the high-level approaches to AI alignment?","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/H9sxfAZGGAsx5BdYD/what-are-the-high-level-approaches-to-ai-alignment",0,"",""],["Causality Adds Up to Normality","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hgSKz3RkSSgZXrXNp/causality-adds-up-to-normality",0,"",""],["dm_control: Software and Tasks for Continuous Control","Yuval Tassa and 9 others","2020","blog","deepmind.com","www.deepmind.com/blog/dm-control-software-and-tasks-for-continuous-control",0,"",""],["Formal Verification of End-to-End Learning in Cyber-Physical Systems: Progress and Challenges","Nathan Fulton and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.09181",0,"","deception agents"],["Modeling the Human Trajectory","David Roodman","2020","report","openphilanthropy.org","www.openphilanthropy.org/blog/modeling-human-trajectory",0,"","robustness"],["Pessimism About Unknown Unknowns Inspires Conservatism","Michael K. Cohen and Marcus Hutter","2020","paper","arXiv preprint","arxiv.org/abs/2006.08753",0,"","agents policy"],["The Social Contract for AI","Mirka Snyder Caron and Abhishek Gupta","2020","paper","arXiv preprint","arxiv.org/abs/2006.08140",0,"",""],["Ensuring safety and consistency in the age of machine learning _ Chongli Qin _ EAGxVirtual 2020-by Centre for Effective Altruism-video_id SS9DMr4VkbY-date 20200615","Chongli Qin","2020","report","drive.google.com","drive.google.com/file/d/19nqPNvLCecVyEk-FFevKhymXYaNH-fpb/view?usp=share_link",0,"",""],["How social science research can inform AI governance _ Baobao Zhang _ EAGxVirtual 2020-by Centre for Effective Altruism-video_id eTkvtHymI9s-date 20200615","Baobao Zhang","2020","report","drive.google.com","drive.google.com/file/d/1MTOo-ntlaB_oAcuBJtQ4c7iKt6LXDlKk/view?usp=share_link",0,"","governance"],["Ethical Considerations for AI Researchers","Kyle Dent","2020","paper","arXiv preprint","arxiv.org/abs/2006.07558",0,"",""],["Online Bayesian Goal Inference for Boundedly-Rational Planning Agents","Tan Zhi-Xuan and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.07532",0,"","agents"],["Preparing for \"The Talk\" with AI projects","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QSBgGv8byWMjmaGE5/preparing-for-the-talk-with-ai-projects",0,"",""],["Open Questions in Creating Safe Open-ended AI: Tensions Between Control and Creativity","Adrien Ecoffet and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.07495",0,"","interpretability benchmarks agents assurance robustness"],["SAMBA: Safe Model-Based & Active Reinforcement Learning","Alexander I. Cowen-Rivers and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.09436",0,"","evals benchmarks"],["Cartesian Boundary as Abstraction Boundary","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/JasCkaPtZEJsYDX8H/cartesian-boundary-as-abstraction-boundary",0,"",""],["Multi-Agent Informational Learning Processes","J. K. Terry and Nathaniel Grammel","2020","paper","arXiv preprint","arxiv.org/abs/2006.06870",0,"","agents"],["[AN #103]: ARCHES: an agenda for existential safety, and combining natural language with deep RL","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gToGqwS9z2QFvwJ7b/an-103-arches-an-agenda-for-existential-safety-and-combining",0,"",""],["What Matters In On-Policy Reinforcement Learning? A Large-Scale Empirical Study","Marcin Andrychowicz and 11 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.05990",0,"","agents policy"],["Contestable Black Boxes","Andrea Aler Tubella and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.05133",0,"","evals assurance"],["June 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/06/08/june-2020-newsletter/",0,"",""],["More on disambiguating \"discontinuity\"","Aryeh Englander","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/C9YMrPAyMXfB8cLPb/more-on-disambiguating-discontinuity",0,"","forecasting"],["Public Static: What is Abstraction?","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/vDGvHBDuMtcPd8Lks/public-static-what-is-abstraction",0,"","mechanistic-interpretability theory"],["Goal-directedness is behavioral, not structural","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9pxcekdNjE7oNwvcC/goal-directedness-is-behavioral-not-structural",0,"",""],["Learning to Play No-Press Diplomacy with Best Response Policy Iteration","Thomas Anthony and 13 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.04635",0,"","agents policy"],["Reinforcement Learning Under Moral Uncertainty","Adrien Ecoffet and Joel Lehman","2020","paper","arXiv preprint","arxiv.org/abs/2006.04734",0,"","agents"],["Curiosity Killed or Incapacitated the Cat and the Asymptotically Optimal Agent","Michael K. Cohen and 2 others","2020","paper","Journal of Selected Areas in Information Theory 2 (2021)","arxiv.org/abs/2006.03357",0,"","agents policy"],["Reply to Paul Christiano on Inaccessible Information","Alex Flint","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/A9vvxguZMytsN3ze9/reply-to-paul-christiano-on-inaccessible-information",0,"",""],["[AN #102]: Meta learning by GPT-3, and a list of full proposals for AI alignment","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/D3hP47pZwXNPRByj8/an-102-meta-learning-by-gpt-3-and-a-list-of-full-proposals",0,"",""],["Focus: you are allowed to be bad at accomplishing your goals","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/X5WTgfX5Ly4ZNHWZD/focus-you-are-allowed-to-be-bad-at-accomplishing-your-goals",0,"","deception policy robustness"],["Inaccessible information","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZyWyAJbedvEgRT2uF/inaccessible-information",0,"",""],["Inaccessible information","Paul Christiano","2020","report","ai-alignment.com","ai-alignment.com/inaccessible-information-c749c6a88ce",0,"",""],["MultiXNet: Multiclass Multistage Multimodal Motion Prediction","Nemanja Djuric and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.02000",0,"","evals"],["AI Definitions Affect Policymaking","Dewey Murdick and 2 others","2020","report","cset.georgetown.edu","cset.georgetown.edu/research/ai-definitions-affect-policymaking/",0,"","policy"],["Aligning Superhuman AI with Human Behavior: Chess as a Model System","Reid McIlroy-Young and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.01855",0,"","agents"],["Building brain-inspired AGI is infinitely easier than understanding the brain","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PTkd8nazvH9HQpwP8/building-brain-inspired-agi-is-infinitely-easier-than",0,"",""],["Acme: A new framework for distributed reinforcement learning","Matt Hoffman and 19 others","2020","blog","deepmind.com","www.deepmind.com/blog/acme-a-new-framework-for-distributed-reinforcement-learning",0,"",""],["Assessing the Risks Posed by the Convergence of Artificial Intelligence and Biotechnology","John T. O'Brien and Cassidy Nelson","2020","report","liebertpub.com","www.liebertpub.com/doi/10.1089/hs.2019.0122",0,"",""],["How to Be Helpful to Multiple People at Once","Vael Gates and 2 others","2020","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/abs/10.1111/cogs.12841",0,"",""],["Medium-Term Artificial Intelligence and Society","Seth D. Baum","2020","report","mdpi.com","www.mdpi.com/2078-2489/11/6/290",0,"",""],["Recordings from AI Safety Discussion Days","AI Safety Support","2020","report","aisafetysupport.org","www.aisafetysupport.org/events/discussion-days#h.hxkmsqyrl6yk",0,"",""],["Shaping the Terrain of AI Competition","Tim Hwang","2020","report","cset.georgetown.edu","cset.georgetown.edu/research/shaping-the-terrain-of-ai-competition/",0,"",""],["Sparsity and interpretability?","Ada Böhm and 2 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/maBNBgopYxb9YZP8B/sparsity-and-interpretability-1",0,"","interpretability agents"],["Word Report #1","Tamsin Leake","2020","blog","carado.moe","carado.moe/word-report-1.html",0,"",""],["Possible takeaways from the coronavirus pandemic for slow AI takeoff","Vika","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/wTKjRFeSjKLDSWyww/possible-takeaways-from-the-coronavirus-pandemic-for-slow-ai",0,"","forecasting"],["Possible takeaways from the coronavirus pandemic for slow AI takeoff","Victoria Krakovna","2020","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2020/05/31/possible-takeaways-from-the-coronavirus-pandemic-for-slow-ai-takeoff/",0,"","forecasting"],["AI Research Considerations for Human Existential Safety (ARCHES)","Andrew Critch and David Krueger","2020","paper","arXiv preprint","arxiv.org/abs/2006.04948",0,"",""],["AISC4: Research Summaries","Sebastian Kosch","2020","blog","aisafety.camp","aisafety.camp/2020/05/30/aisc4-research-summaries/",0,"",""],["An overview of 11 proposals for building safe advanced AI","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fRsjBseRuvRhMPPE5/an-overview-of-11-proposals-for-building-safe-advanced-ai",0,"","scalable-oversight debate interpretability evals agents policy"],["GPT-3: a disappointing paper","nostalgebraist","2020","blog","LessWrong","www.lesswrong.com/posts/ZHrpjDc3CepSeeBuE/gpt-3-a-disappointing-paper",0,"","forecasting"],["May 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/05/29/may-2020-newsletter/",0,"",""],["Language Models are Few-Shot Learners","Tom B. Brown and 30 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.14165",0,"",""],["[AN #101]: Why we should rigorously measure and forecast AI progress","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/axzPYvcmWr2TwvnLi/an-101-why-we-should-rigorously-measure-and-forecast-ai",0,"","forecasting"],["AI Forensics: Did the Artificial Intelligence System Do It? Why?","Johannes Schneider and Frank Breitinger","2020","paper","arXiv preprint","arxiv.org/abs/2005.13635",0,"","governance"],["AI Safety Discussion Days","Linda Linsefors","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/32QD3tRfognNHN9xw/ai-safety-discussion-days",0,"",""],["The Adversarial Resilience Learning Architecture for AI-based Modelling, Exploration, and Operation of Complex Cyber-Physical Systems","Eric MSP Veith and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.13601",0,"","agents"],["How can Interpretability help Alignment?","RobertKirk and 2 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/uRnprGSiLGXv35foX/how-can-interpretability-help-alignment",0,"","scalable-oversight debate interpretability"],["AGIs as collectives","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/HekjhtWesBWTQW5eF/agis-as-collectives",0,"",""],["Danny Hernandez on forecasting and the drivers of AI progress","Robert Wiblin and 2 others","2020","report","80000hours.org","80000hours.org/podcast/episodes/danny-hernandez-forecasting-ai-progress/",0,"","forecasting"],["From ImageNet to Image Classification: Contextualizing Progress on Benchmarks","Dimitris Tsipras and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.11295",0,"","evals benchmarks"],["AI Research Considerations for Human Existential Safety (ARCHES)","Andrew Critch","2020","blog","EA Forum","forum.effectivealtruism.org/posts/aYg2ceChLMRbwqkyQ/ai-research-considerations-for-human-existential-safety",0,"",""],["Comparing reward learning/reward tampering formalisms","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/MBrhMSZno6qbfGQdZ/comparing-reward-learning-reward-tampering-formalisms",0,"","reward-hacking"],["[AN #100]: What might go wrong if you learn a reward function while acting","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/GYmDaFgePMchYj6P7/an-100-what-might-go-wrong-if-you-learn-a-reward-function",0,"",""],["Probabilities, weights, sums: pretty much the same for reward functions","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pxWEKPHNBzXZWi2rB/probabilities-weights-sums-pretty-much-the-same-for-reward",0,"",""],["Rational Consensus","Joseph Y. Halpern and Xavier Vilaca","2020","paper","arXiv preprint","arxiv.org/abs/2005.10141",0,"","agents"],["What Makes for Good Views for Contrastive Learning?","Yonglong Tian and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.10243",0,"","robustness"],["Human Instruction-Following with Deep Reinforcement Learning via Transfer-Learning from Text","Felix Hill and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.09382",0,"","agents policy"],["Learning and manipulating learning","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LpjjWDBXr88gzcYK2/learning-and-manipulating-learning",0,"",""],["Pointing to a Flower","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3xotPYdAs7GfT9a9r/pointing-to-a-flower",0,"",""],["Reward functions and updating assumptions can hide a multitude of sins","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/EYEkYX6vijL7zsKEt/reward-functions-and-updating-assumptions-can-hide-a",0,"",""],["The Mechanistic and Normative Structure of Agency","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QBHxfATzdASQcXwan/the-mechanistic-and-normative-structure-of-agency",0,"",""],["Why you should minimax in two-player zero-sum games","Nisan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ukZuzb8JpYiFLoord/why-you-should-minimax-in-two-player-zero-sum-games",0,"",""],["AI Governance Career Paths for Europeans","careersthrowaway","2020","blog","EA Forum","forum.effectivealtruism.org/posts/WqQaPYhzDYJwLC6gW/ai-governance-career-paths-for-europeans",0,"","governance policy"],["Multi-agent safety","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BXMCgpktdiawT3K5v/multi-agent-safety",0,"","agents"],["Conjecture Workshop","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/uDmiEvPtJnRcrbHB6/conjecture-workshop",0,"",""],["How should AIs update a prior over human preferences?","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gbuwgyYG9WvtsErki/how-should-ais-update-a-prior-over-human-preferences",0,"",""],["Language Conditioned Imitation Learning over Unstructured Data","Corey Lynch and Pierre Sermanet","2020","paper","arXiv preprint","arxiv.org/abs/2005.07648",0,"","agents policy"],["Overcoming Barriers to Cross-cultural Cooperation in AI Ethics and Governance","Seán S. ÓhÉigeartaigh and 4 others","2020","report","link.springer.com","link.springer.com/10.1007/s13347-020-00402-x",0,"","governance"],["[AN #99]: Doubling times for the efficiency of AI algorithms","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/R6gPKJAq6dbuLNkwG/an-99-doubling-times-for-the-efficiency-of-ai-algorithms",0,"",""],["GovAI Webinars on the Governance and Economics of AI","MarkusAnderljung","2020","blog","EA Forum","forum.effectivealtruism.org/posts/RKy2emQdEgQgqv5ok/govai-webinars-on-the-governance-and-economics-of-ai",0,"","governance"],["Planning to Explore via Self-Supervised World Models","Ramanan Sekar and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.05960",0,"","evals agents"],["Simple Sensor Intentions for Exploration","Tim Hertweck and 7 others","2020","blog","deepmind.com","www.deepmind.com/blog/simple-sensor-intentions-for-exploration",0,"",""],["Book report: Theory of Games and Economic Behavior (von Neumann & Morgenstern)","Nisan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qRKyZGcoio9JhdmvX/book-report-theory-of-games-and-economic-behavior-von",0,"",""],["Critical Review of 'The Precipice': A Reassessment of the Risks of AI and Pandemics","Fods12","2020","blog","EA Forum","forum.effectivealtruism.org/posts/2sMR7n32FSvLCoJLQ/critical-review-of-the-precipice-a-reassessment-of-the-risks",0,"",""],["Corrigibility as outside view","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BMj6uMuyBidrdZkiD/corrigibility-as-outside-view",0,"","agents robustness"],["Measuring the Algorithmic Efficiency of Neural Networks","Danny Hernandez and Tom B. Brown","2020","paper","arXiv preprint","arxiv.org/abs/2005.04305",0,"","robustness"],["A Reinforcement Learning Potpourri","Alex Irpan","2020","report","alexirpan.com","www.alexirpan.com/2020/05/07/rl-potpourri.html",0,"",""],["Learning to Segment Actions from Observation and Narration","Daniel Fried* and 5 others","2020","blog","deepmind.com","www.deepmind.com/blog/learning-to-segment-actions-from-observation-and-narration",0,"",""],["[AN #98]: Understanding neural net training by seeing which gradients were helpful","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Sj9YurD9vwpfPErs2/an-98-understanding-neural-net-training-by-seeing-which",0,"",""],["Maths writer/cowritter needed: how you can't distinguish early exponential from early sigmoid","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zCq4ca3tTcfQgrFZM/maths-writer-cowritter-needed-how-you-can-t-distinguish",0,"",""],["Modeling naturalized decision problems in linear logic","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bcFhPHcDRbWKcAEfk/modeling-naturalized-decision-problems-in-linear-logic",0,"","theory"],["Specification gaming: the flip side of AI ingenuity","Vika and 5 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/7b2RJJQ76hjZwarnj/specification-gaming-the-flip-side-of-ai-ingenuity",0,"","specification-gaming goodharts-law"],["Specification gaming: the flip side of AI ingenuity","Vika and 5 others","2020","blog","LessWrong","www.lesswrong.com/posts/7b2RJJQ76hjZwarnj/specification-gaming-the-flip-side-of-ai-ingenuity",0,"","specification-gaming goodharts-law"],["A multi-component framework for the analysis and design of explainable artificial intelligence","S. Atakishiyev and 8 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.01908",0,"","interpretability"],["Competitive safety via gradated curricula","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/vLepnCxCWW6YTw8eW/competitive-safety-via-gradated-curricula",0,"",""],["Exploring Bayesian Optimization","Apoorv Agnihotri and Nipun Batra","2020","report","Distill","distill.pub/2020/bayesian-optimization",0,"",""],["Writing Causal Models Like We Write Programs","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Xd9FLs4geRAWxkQPE/writing-causal-models-like-we-write-programs",0,"",""],["Evaluating Explainable AI: Which Algorithmic Explanations Help Users Predict Model Behavior?","Peter Hase and Mohit Bansal","2020","paper","arXiv preprint","arxiv.org/abs/2005.01831",0,"","interpretability evals"],["How uniform is the neocortex?","zhukeepa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WFopenhCXyHX3ukw3/how-uniform-is-the-neocortex",0,"",""],["Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","Sergey Levine and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.01643",0,"",""],["Open Loop In Natura Economic Planning","Spyridon Samothrakis","2020","paper","arXiv preprint","arxiv.org/abs/2005.01539",0,"",""],["\"Don't even think about hell\"","emmab","2020","blog","LessWrong","www.lesswrong.com/posts/hGmFNBXDinfiKJGD6/don-t-even-think-about-hell",0,"","theory"],["How does iterated amplification exceed human abilities?","riceissa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ajQzejMYizfX4dMWK/how-does-iterated-amplification-exceed-human-abilities",0,"","scalable-oversight"],["Stanford Encyclopedia of Philosophy on AI ethics and superintelligence","Kaj_Sotala","2020","blog","LessWrong","www.lesswrong.com/posts/gmsAWkcQRJst2Jnrk/stanford-encyclopedia-of-philosophy-on-ai-ethics-and",0,"","forecasting"],["April 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/05/01/april-2020-newsletter/",0,"",""],["Learning to Complement Humans","Bryan Wilder and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2005.00582",0,"",""],["Topological metaphysics: relating point-set topology and locale theory","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/yTvZFzcgt7rGYMxP5/topological-metaphysics-relating-point-set-topology-and",0,"",""],["Optimising Society to Constrain Risk of War from an Artificial Superintelligence","JohnCDraper","2020","blog","LessWrong","www.lesswrong.com/posts/cpGzrF7XztMjzzD2H/optimising-society-to-constrain-risk-of-war-from-an",0,"","governance"],["Reinforcement Learning with Augmented Data","Michael Laskin and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.14990",0,"","robustness"],["What is the alternative to intent alignment called?","Richard_Ngo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/NuhsBLxxswinm2JKZ/what-is-the-alternative-to-intent-alignment-called",0,"",""],["[AN #97]: Are there historical examples of large, robust discontinuities?","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WknLjywekGajwD2fp/an-97-are-there-historical-examples-of-large-robust",0,"","forecasting"],["Motivating Abstraction-First Decision Theory","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/oQHoy2tnLsKuEDYtJ/motivating-abstraction-first-decision-theory",0,"","theory"],["the Economic Compass","Tamsin Leake","2020","blog","carado.moe","carado.moe/economic-compass.html",0,"",""],["Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","Ilya Kostrikov and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.13649",0,"",""],["Pitfalls of learning a reward function online","Stuart Armstrong and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.13654",0,"","agents"],["Is the Most Accurate AI the Best Teammate? Optimizing AI for Teamwork","Gagan Bansal and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.13102",0,"",""],["MIRI’s largest grant to date!","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/04/27/miris-largest-grant-to-date/",0,"",""],["Causal Mediation Analysis for Interpreting Neural NLP: The Case of Gender Bias","Jesse Vig","2020","paper","arXiv preprint","arxiv.org/abs/2004.12265",0,"",""],["Limiting Real Universes","Tamsin Leake","2020","blog","carado.moe","carado.moe/limiting-real-universes.html",0,"",""],["Fast Takeoff in Biological Intelligence","anonymous","2020","blog","LessWrong","www.lesswrong.com/posts/eweeg8iHX5SK5oHvs/fast-takeoff-in-biological-intelligence",0,"","forecasting"],["What are the relative speeds of AI capabilities and AI safety?","NunoSempere","2020","blog","LessWrong","www.lesswrong.com/posts/EbX5gv62oyyWZgwZm/what-are-the-relative-speeds-of-ai-capabilities-and-ai",0,"",""],["What makes counterfactuals comparable?","Chris_Leong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6E6D3qLPM3urXDPpK/what-makes-counterfactuals-comparable-1",0,"","theory"],["DeepMind team on specification gaming","JoshuaFox","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZEoyoccFoBQQRzbz2/deepmind-team-on-specification-gaming",0,"","specification-gaming"],["Responsible AI and Its Stakeholders","Gabriel Lima and Meeyoung Cha","2020","paper","arXiv preprint","arxiv.org/abs/2004.11434",0,"",""],["[AN #96]: Buck and I discuss/argue about AI Alignment","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YyKKMeCCxnzdohuxj/an-96-buck-and-i-discuss-argue-about-ai-alignment",0,"",""],["A Neural Scaling Law from the Dimension of the Data Manifold","Utkarsh Sharma and Jared Kaplan","2020","paper","arXiv preprint","arxiv.org/abs/2004.10802",0,"","scaling-laws"],["Description vs simulated prediction","richardkorzekwa","2020","blog","aiimpacts.org","aiimpacts.org/description-vs-simulated-prediction/",0,"",""],["Inner alignment in the brain","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/DWFx2Cmsvd4uCKkZ4/inner-alignment-in-the-brain",0,"",""],["Problem relaxation as a tactic","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/JcpwEKbmNHdwhpq5n/problem-relaxation-as-a-tactic",0,"",""],["BERT-ATTACK: Adversarial Attack Against BERT Using BERT","Linyang Li and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.09984",0,"",""],["Databases of human behaviour and preferences?","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fx8Mdorwmt696Ramm/databases-of-human-behaviour-and-preferences",0,"",""],["AI Services as a Research Paradigm","VojtaKovarik","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/z2ofM2oZQwmcWFt8N/ai-services-as-a-research-paradigm",0,"","agents"],["Dark, Beyond Deep: A Paradigm Shift to Cognitive AI with Humanlike Common Sense","Yixin Zhu and 11 others","2020","paper","Engineering, Feb, 2020","arxiv.org/abs/2004.09044",0,"","agents robustness"],["Intuitions on Universal Behavior of Information at a Distance","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/CxEbvETK2WNfHw7v9/intuitions-on-universal-behavior-of-information-at-a",0,"",""],["How do you talk about AI safety?","BrownHairedEevee","2020","blog","EA Forum","forum.effectivealtruism.org/posts/pJFSKMdq8MvuQzAsw/how-do-you-talk-about-ai-safety",0,"",""],["Three Modern Roles for Logic in AI","Adnan Darwiche","2020","paper","arXiv preprint","arxiv.org/abs/2004.08599",0,"","mechanistic-interpretability"],["Discontinuous progress in history: an update","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/5SYG9tjv2E4kyE9Zi/discontinuous-progress-in-history-an-update",0,"","forecasting"],["AI Alignment Podcast: An Overview of Technical AI Alignment in 2018 and 2019 with Buck Shlegeris and Rohin Shah","Palus Astra","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6skeZgctugzBBEBw3/ai-alignment-podcast-an-overview-of-technical-ai-alignment",0,"","governance"],["Improving Verifiability in AI Development","OpenAI","2020","report","openai.com","openai.com/blog/improving-verifiability/",0,"","interpretability red-teaming governance"],["Integrating Hidden Variables Improves Approximation","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4vrL94CqXuyHQMhqo/integrating-hidden-variables-improves-approximation",0,"",""],["Subjectifying Objectivity: Delineating Tastes in Theoretical Quantum Gravity Research","Thomas K. Gilbert and Andrew J. Loveridge","2020","paper","arXiv preprint","arxiv.org/abs/2004.07450",0,"",""],["[AN #95]: A framework for thinking about how to make AI go well","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9et86yPRk6RinJNt3/an-95-a-framework-for-thinking-about-how-to-make-ai-go-well",0,"",""],["A Collection Of Compasses","Tamsin Leake","2020","blog","carado.moe","carado.moe/compasses.html",0,"",""],["Aligning AI to Human Values means Picking the Right Metrics","Jonathan Stray","2020","report","medium.com","medium.com/partnership-on-ai/aligning-ai-to-human-values-means-picking-the-right-metrics-855859e6f047",0,"",""],["Database of existential risk estimates","MichaelA","2020","blog","EA Forum","forum.effectivealtruism.org/posts/JQQAQrunyGGhzE23a/database-of-existential-risk-estimates",0,"","forecasting"],["Toward Trustworthy AI Development: Mechanisms for Supporting Verifiable Claims","Miles Brundage and 39 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.07213",0,"","governance"],["2019 recent trends in Geekbench score per CPU price","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/2019-recent-trends-in-geekbench-score-per-cpu-price/",0,"",""],["Discontinuous progress in history: an update","KatjaGrace","2020","blog","LessWrong","www.lesswrong.com/posts/CeZXDmp8Z363XaM6b/discontinuous-progress-in-history-an-update",0,"","forecasting"],["Precedents for economic n-year doubling before 4n-year doubling","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/precedents-for-economic-n-year-doubling-before-4n-year-doubling/",0,"",""],["Resolutions of mathematical conjectures over time","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/resolutions-of-mathematical-conjectures-over-time/",0,"",""],["Surveys on fractional progress towards HLAI","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/surveys-on-fractional-progress-towards-hlai/",0,"","forecasting"],["Trends in DRAM price per gigabyte","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/trends-in-dram-price-per-gigabyte/",0,"",""],["\"How conservative\" should the partial maximisers be?","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/jRHLCyRKsQv5u2Lph/how-conservative-should-the-partial-maximisers-be",0,"",""],["Discontinuous progress in history: an update","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/discontinuous-progress-in-history-an-update/",0,"","benchmarks forecasting"],["Certifiable Robustness to Adversarial State Uncertainty in Deep Reinforcement Learning","Michael Everett and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.06496",0,"","policy robustness"],["Asymptotically Unambitious AGI","michaelcohen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pZhDWxDmwzuSwLjou/asymptotically-unambitious-agi",0,"","instrumental-convergence"],["[AN #94]: AI alignment as translation between humans and machines","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gor57NZtxG4bq5eej/an-94-ai-alignment-as-translation-between-humans-and",0,"",""],["CURL: Contrastive Unsupervised Representations for Reinforcement Learning","Aravind Srinivas and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.04136",0,"","benchmarks policy"],["An Orthodox Case Against Utility Functions","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/A8iGaZ3uHNNGgJeaD/an-orthodox-case-against-utility-functions",0,"","theory"],["Takeaways from safety by default interviews","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/hJDSSTMcv9teNfHQM/takeaways-from-safety-by-default-interviews",0,"",""],["TuringAdvice: A Generative and Dynamic Evaluation of Language Use","Rowan Zellers and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.03607",0,"","evals robustness"],["Announcing Web-TAISU, May 13-17","Linda Linsefors","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/CMnMaTxNAhXfcEtgm/announcing-web-taisu-may-13-17",0,"",""],["Preliminary survey of prescient actions","richardkorzekwa","2020","blog","aiimpacts.org","aiimpacts.org/survey-of-prescient-actions/",0,"",""],["Resources for AI Alignment Cartography","Gyrodiot","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4az2cFrJp3ya4y6Wx/resources-for-ai-alignment-cartography",0,"",""],["Paul Christiano: Current work in AI alignment","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/63stBTw3WAW6k45dY/paul-christiano-current-work-in-ai-alignment",0,"","scalable-oversight agents robustness"],["Robots Learning to Move like Animals","Daniel Seita","2020","report","bair.berkeley.edu","bair.berkeley.edu/blog/2020/04/03/laikago/",0,"",""],["Takeaways from safety by default interviews","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/takeaways-from-safety-by-default-interviews/",0,"",""],["AGI in a vulnerable world","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/RCfSmyGwyyvDFGYqL/agi-in-a-vulnerable-world",0,"",""],["Atari early","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/atari-early/",0,"","agents"],["Atari early","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/R9XhTTyrNQR8PvsRf/atari-early",0,"","forecasting"],["Atari early","KatjaGrace","2020","blog","LessWrong","www.lesswrong.com/posts/ygb6ryKcScJxhmwQo/atari-early",0,"","forecasting"],["Equilibrium and prior selection problems in multipolar deployment","JesseClifton","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Tdu3tGT4i24qcLESh/equilibrium-and-prior-selection-problems-in-multipolar-1",0,"","agents governance"],["Equilibrium and prior selection problems in multipolar deployment","JesseClifton","2020","blog","LessWrong","www.lesswrong.com/posts/Tdu3tGT4i24qcLESh/equilibrium-and-prior-selection-problems-in-multipolar-1",0,"","governance"],["Interviews on plausibility of AI safety by default","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/interviews-on-plausibility-of-ai-safety-by-default/",0,"",""],["Three kinds of competitiveness","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/zpaz2n5pT4xLeF3K9/three-kinds-of-competitiveness",0,"",""],["What is the subjective experience of free will for agents?","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BTM4SN53mWsHLkRJL/what-is-the-subjective-experience-of-free-will-for-agents",0,"","agents theory"],["[AN #93]: The Precipice we’re standing at, and how we can back away from it","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/rPC9Y9b5vkTqakywC/an-93-the-precipice-we-re-standing-at-and-how-we-can-back",0,"",""],["AI Research with the Potential for Malicious Use: Publication Norms and Governance Considerations","Seán Ó hÉigeartaigh","2020","report","aigovernancereview.com","www.aigovernancereview.com/static/AI-Governance-in-2020-ffa2e9c4e0ec4ca3706455e0f35d5ab5.pdf",0,"","governance"],["An Overview of Early Vision in InceptionV1","Chris Olah and 5 others","2020","report","Distill","distill.pub/2020/circuits/early-vision",0,"",""],["CHAI Newsletter #1 2020","CHAI","2020","report","drive.google.com","drive.google.com/file/d/16bG2w-k3wPP7ZOPrqQJIyW094BkmBwXR/view?usp=sharing",0,"",""],["Counterfactual Multi-Agent Reinforcement Learning with Graph Convolution Communication","Jianyu Su and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2004.00470",0,"","interpretability evals agents policy"],["How special are human brains among animal brains?","zhukeepa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/d2jgBurQygbXzhPxc/how-special-are-human-brains-among-animal-brains",0,"","forecasting"],["How special are human brains among animal brains?","zhukeepa","2020","blog","LessWrong","www.lesswrong.com/posts/d2jgBurQygbXzhPxc/how-special-are-human-brains-among-animal-brains",0,"","forecasting"],["March 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/04/01/march-2020-newsletter/",0,"",""],["Meta-preferences two ways: generator vs. patch","Charlie Steiner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/A5jN7vqAxsHCDC4dy/meta-preferences-two-ways-generator-vs-patch",0,"",""],["Two Alternatives to Logical Counterfactuals","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/yBdDXXmLYejrcPPv2/two-alternatives-to-logical-counterfactuals",0,"","theory"],["What achievements have people claimed will be warning signs for AGI?","Richard_Ngo","2020","blog","LessWrong","www.lesswrong.com/posts/JHpdpZYnb6hQdR82f/what-achievements-have-people-claimed-will-be-warning-signs",0,"","forecasting"],["AI Services: Introduction v1.3","Vojta Kovarik","2020","report","docs.google.com","docs.google.com/document/d/1SYgvWBe1ruDl9dQnxmjll-8COUHPycGOlLvTI68xtLA/edit?pli=1&usp=embed_facebook",0,"",""],["Outperforming the human Atari benchmark","Vaniver","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/kwgM5nGe9QXcB4TTu/outperforming-the-human-atari-benchmark",0,"","benchmarks"],["Three Kinds of Competitiveness","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/sD6KuprcS3PFym2eM/three-kinds-of-competitiveness",0,"",""],["Three kinds of competitiveness","Daniel Kokotajlo","2020","blog","aiimpacts.org","aiimpacts.org/three-kinds-of-competitiveness/",0,"",""],["Agent57: Outperforming the Atari Human Benchmark","Adrià Puigdomènech Badia and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.13350",0,"","benchmarks agents policy robustness"],["Book Review: 12 Rules For Life","Tamsin Leake","2020","blog","carado.moe","carado.moe/12-rules-for-life.html",0,"",""],["My current framework for thinking about AGI timelines","zhukeepa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/w4jjwDPa853m9P4ag/my-current-framework-for-thinking-about-agi-timelines",0,"","forecasting"],["Suphx: Mastering Mahjong with Deep Reinforcement Learning","Junjie Li and 9 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.13590",0,"","evals agents policy"],["the Belief In Society compass","Tamsin Leake","2020","blog","carado.moe","carado.moe/belief-in-society.html",0,"",""],["Topia: Layer 0","Tamsin Leake","2020","blog","carado.moe","carado.moe/topia-layer-0.html",0,"",""],["On Economics","Tamsin Leake","2020","blog","carado.moe","carado.moe/on-economics.html",0,"",""],["AGI in a vulnerable world","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/agi-in-a-vulnerable-world/",0,"",""],["How important are MDPs for AGI (Safety)?","michaelcohen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6gL83HMF6tvPHKQxW/how-important-are-mdps-for-agi-safety",0,"",""],["What are the most plausible \"AI Safety warning shot\" scenarios?","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/hLKKH9CM6NDiJBabC/what-are-the-most-plausible-ai-safety-warning-shot-scenarios",0,"",""],["2019 recent trends in GPU price per FLOPS","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/2019-recent-trends-in-gpu-price-per-flops/",0,"",""],["[AN #92]: Learning good representations with contrastive predictive coding","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XE6LD2c9NtB7gMdEm/an-92-learning-good-representations-with-contrastive",0,"","robustness"],["An empirical investigation of the challenges of real-world reinforcement learning","Gabriel Dulac-Arnold and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.11881",0,"","benchmarks"],["The Precipice: Existential Risk and the Future of Humanity","Toby Ord","2020","report","goodreads.com","www.goodreads.com/book/show/48570420-the-precipice",0,"",""],["Deconfusing Human Values Research Agenda v1","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/k8F8TBzuZtLheJt47/deconfusing-human-values-research-agenda-v1",0,"",""],["[Meta] Do you want AIS Webinars?","Linda Linsefors","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/BbrsgHPJmGxeg7nXG/meta-do-you-want-ais-webinars",0,"",""],["Fireside chat - AI governance _ Markus Anderljung _ Ben Garfinkel _ EA Global - Virtual 2020-by Centre for Effective Altruism-video_id bSTYiIgjgrk-date 20200321","Markus Anderljung and Ben Garfinkel","2020","report","drive.google.com","drive.google.com/file/d/1-F1AubRvkU95p8y5qUIye44WvwQg-Zid/view?usp=share_link",0,"","governance"],["Mediation From a Distance","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/DqfpcFwfeZHFe5J8h/mediation-from-a-distance",0,"",""],["Rohin Shah_ WhatΓÇÖs been happening in AI alignment_-by EA Global Virtual 2020-date 20200321","Rohin Shah","2020","report","drive.google.com","drive.google.com/file/d/1NbBk8tN9hxClfoScoHGzk7M5iEvktGIH/view?usp=share_link",0,"",""],["Abstraction = Information at a Distance","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TTNS3tk5McHqrJCbR/abstraction-information-at-a-distance",0,"",""],["Alignment as Translation","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/42YykiTqtGMyJAjDM/alignment-as-translation",0,"","robustness"],["Thinking About Filtered Evidence Is (Very!) Hard","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fhJkQo34cYw6KqpH3/thinking-about-filtered-evidence-is-very-hard",0,"",""],["[AN #91]: Concepts, implementations, problems, and a benchmark for impact measurement","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dJanptWZnZx5omwBz/an-91-concepts-implementations-problems-and-a-benchmark-for",0,"","benchmarks"],["Proxy tasks and subjective measures can be misleading in evaluating explainable AI systems","Zana Buçinca and 3 others","2020","report","doi.org","doi.org/10.1145/3377325.3377498",0,"","evals"],["What is Interpretability?","RobertKirk and 2 others","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/rSMbGFfsLMB3GWZtX/what-is-interpretability",0,"","interpretability"],["AI Alignment Podcast: On Lethal Autonomous Weapons with Paul Scharre","Palus Astra","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/xEzudcydk7APZbnai/ai-alignment-podcast-on-lethal-autonomous-weapons-with-paul",0,"","robustness"],["DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","Aviral Kumar and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.07305",0,"","agents policy"],["Visualizing Neural Networks with the Grand Tour","Mingwei Li and 2 others","2020","report","Distill","distill.pub/2020/grand-tour",0,"",""],["What are some exercises for building/generating intuitions about key disagreements in AI alignment?","riceissa","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bDwQddhqaTiMhbpPF/what-are-some-exercises-for-building-generating-intuitions",0,"",""],["[Link and commentary] Beyond Near- and Long-Term: Towards a Clearer Account of Research Priorities in AI Ethics and Society","MichaelA","2020","blog","EA Forum","forum.effectivealtruism.org/posts/jyaRY8yWgv679XS7p/link-and-commentary-beyond-near-and-long-term-towards-a",0,"","governance"],["Fast and Easy Infinitely Wide Networks with Neural Tangents","Samuel S Schoenholz and Roman Novak","2020","report","ai.googleblog.com","ai.googleblog.com/2020/03/fast-and-easy-infinitely-wide-networks.html",0,"",""],["The Conflict Between People's Urge to Punish AI and Legal Systems","Gabriel Lima and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.06507",0,"","agents robustness"],["Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","Christian Scheller and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.06066",0,"","agents policy"],["[AN #90]: How search landscapes can contain self-reinforcing feedback loops","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/7d2PsdHXrJnbofrvF/an-90-how-search-landscapes-can-contain-self-reinforcing",0,"",""],["Trace README","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/DbWoZNxgwr2NBFdoo/trace-readme",0,"",""],["Visual Grounding in Video for Unsupervised Word Translation","Gunnar Sigurdsson* and 7 others","2020","blog","deepmind.com","www.deepmind.com/blog/visual-grounding-in-video-for-unsupervised-word-translation",0,"",""],["Curriculum Learning for Reinforcement Learning Domains: A Framework and Survey","Sanmit Narvekar and 5 others","2020","paper","Journal of Machine Learning Research 21(181):1-50, 2020","arxiv.org/abs/2003.04960",0,"","agents"],["Pruned Neural Networks are Surprisingly Modular","Daniel Filan and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.04881",0,"","interpretability"],["Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","Stephanie Milani and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.05012",0,"",""],["Thread: Circuits","Nick Cammarata and 39 others","2020","report","Distill","distill.pub/2020/circuits",0,"","mechanistic-interpretability"],["Zoom In: An Introduction to Circuits","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/MG4ZjWQDrdpgeu8wG/zoom-in-an-introduction-to-circuits",0,"","mechanistic-interpretability"],["Zoom In: An Introduction to Circuits","Chris Olah and 5 others","2020","report","Distill","distill.pub/2020/circuits/zoom-in",0,"","mechanistic-interpretability"],["Improved Baselines with Momentum Contrastive Learning","Xinlei Chen and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.04297",0,"",""],["\"Other-Play\" for Zero-Shot Coordination","Hengyuan Hu and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.02979",0,"","deception agents"],["AutoML-Zero: Evolving Machine Learning Algorithms From Scratch","Esteban Real and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.03384",0,"",""],["Can ML predict the solution value for a difficult combinatorial problem?","Constantine Goulimis and Gastón Simone","2020","paper","arXiv preprint","arxiv.org/abs/2003.03181",0,"",""],["A critical agential account of free will, causation, and physics","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dvaCebTNc2tfMDcxS/a-critical-agential-account-of-free-will-causation-and",0,"","agents"],["A critical agential account of free will, causation, and physics","Jessica Taylor","2020","report","unstableontology.com","unstableontology.com/2020/03/05/a-critical-agential-account-of-free-will-causation-and-physics/",0,"","agents"],["[AN #89]: A unifying formalism for preference learning algorithms","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/7bNXqdDPYpnfCNQhA/an-89-a-unifying-formalism-for-preference-learning",0,"",""],["Marketplace for AI Models","Abhishek Kumar and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.01593",0,"",""],["Reliable evaluation of adversarial robustness with an ensemble of diverse parameter-free attacks","Francesco Croce and Matthias Hein","2020","paper","arXiv preprint","arxiv.org/abs/2003.01690",0,"","evals robustness"],["Two Decades of AI4NETS-AI/ML for Data Networks: Challenges & Research Directions","Pedro Casas","2020","paper","5th IEEE/IFIP International Workshop on Analytics for Network and\n  Service Management (AnNet 2020)","arxiv.org/abs/2003.04080",0,"","deception agents"],["Anthropics over-simplified: it's about priors, not updates","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Hpam4RrJKfufXrmAi/anthropics-over-simplified-it-s-about-priors-not-updates",0,"",""],["Cortés, Pizarro, and Afonso as Precedents for Takeover","AI Impacts","2020","blog","EA Forum","forum.effectivealtruism.org/posts/MNPrXCsPpwTgygMxc/cortes-pizarro-and-afonso-as-precedents-for-takeover",0,"",""],["If I were a well-intentioned AI... IV: Mesa-optimising","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/aqhMLqaoHb7uob7fr/if-i-were-a-well-intentioned-ai-iv-mesa-optimising",0,"",""],["If I were a well-intentioned AI... IV: Mesa-optimising","Stuart_Armstrong","2020","blog","LessWrong","www.lesswrong.com/posts/aqhMLqaoHb7uob7fr/if-i-were-a-well-intentioned-ai-iv-mesa-optimising",0,"",""],["An Analytic Perspective on AI Alignment","DanielFilan","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8GdPargak863xaebm/an-analytic-perspective-on-ai-alignment",0,"","interpretability"],["Cooperation, Conflict, and Transformative Artificial Intelligence - A Research Agenda","Jesse Clifton","2020","report","longtermrisk.org","longtermrisk.org/files/Cooperation-Conflict-and-Transformative-Artificial-Intelligence-A-Research-Agenda.pdf",0,"",""],["Cortés, Pizarro, and Afonso as Precedents for Takeover","Daniel Kokotajlo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ivpKSjM4D6FbqF4pZ/cortes-pizarro-and-afonso-as-precedents-for-takeover",0,"",""],["Cortés, Pizarro, and Afonso as precedents for takeover","Daniel Kokotajlo","2020","blog","aiimpacts.org","aiimpacts.org/cortes-pizarro-and-afonso-as-precedents-for-ai-takeover/",0,"",""],["My Updating Thoughts on AI policy","Ben Pace","2020","blog","LessWrong","www.lesswrong.com/posts/DhuPBkKA8ohyZEN8n/my-updating-thoughts-on-ai-policy",0,"","governance policy"],["Responsible AI—Two Frameworks for Ethical Design Practice","Dorian Peters and 3 others","2020","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/9001063/",0,"",""],["Social choice ethics in artificial intelligence","Seth D. Baum","2020","report","link.springer.com","link.springer.com/10.1007/s00146-017-0760-1",0,"",""],["On Safety Assessment of Artificial Intelligence","Jens Braband and Hendrik Schäbe","2020","paper","Dependability, vol. 20 no. 4, 2020","arxiv.org/abs/2003.00260",0,"",""],["Conclusion to 'Reframing Impact'","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/sHpiiZS2gPgoPnijX/conclusion-to-reframing-impact",0,"",""],["Efficiently Guiding Imitation Learning Agents with Human Gaze","Akanksha Saran and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.12500",0,"","agents policy"],["If I were a well-intentioned AI... III: Extremal Goodhart","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/NdJtfujX4sE6xLCsb/if-i-were-a-well-intentioned-ai-iii-extremal-goodhart",0,"","goodharts-law"],["If I were a well-intentioned AI... III: Extremal Goodhart","Stuart_Armstrong","2020","blog","LessWrong","www.lesswrong.com/posts/NdJtfujX4sE6xLCsb/if-i-were-a-well-intentioned-ai-iii-extremal-goodhart",0,"","goodharts-law"],["On Catastrophic Interference in Atari 2600 Games","William Fedus and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.12499",0,"","policy"],["Trace: Goals and Principles","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/rt5X74Az3mXwTubRA/trace-goals-and-principles",0,"",""],["[AN #88]: How the principal-agent literature relates to AI risk","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/y9JeNZ2WAkR6MbBZH/an-88-how-the-principal-agent-literature-relates-to-ai-risk",0,"","agents"],["Attainable Utility Preservation: Scaling to Superhuman","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/S8AGyJJsdBFXmxHcb/attainable-utility-preservation-scaling-to-superhuman",0,"",""],["If I were a well-intentioned AI... II: Acting in a world","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZKzAjKSeNRtiaeJns/if-i-were-a-well-intentioned-ai-ii-acting-in-a-world",0,"",""],["If I were a well-intentioned AI... II: Acting in a world","Stuart_Armstrong","2020","blog","LessWrong","www.lesswrong.com/posts/ZKzAjKSeNRtiaeJns/if-i-were-a-well-intentioned-ai-ii-acting-in-a-world",0,"",""],["Reasons for Excitement about Impact of Impact Measure Research","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/wAAvP8RG6EwzCvHJy/reasons-for-excitement-about-impact-of-impact-measure",0,"",""],["State-only Imitation with Transition Dynamics Mismatch","Tanmay Gangwani and Jian Peng","2020","paper","arXiv preprint","arxiv.org/abs/2002.11879",0,"","agents policy"],["Cautious Reinforcement Learning with Logical Constraints","Mohammadhosein Hasanbeig and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.12156",0,"","agents"],["Generalized Hindsight for Reinforcement Learning","Alexander C. Li and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.11708",0,"","policy"],["If I were a well-intentioned AI... I: Image classifier","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gzWb5kWwzhdaqmyTt/if-i-were-a-well-intentioned-ai-i-image-classifier",0,"","robustness"],["If I were a well-intentioned AI... I: Image classifier","Stuart_Armstrong","2020","blog","LessWrong","www.lesswrong.com/posts/gzWb5kWwzhdaqmyTt/if-i-were-a-well-intentioned-ai-i-image-classifier",0,"","robustness"],["Rethinking Bias-Variance Trade-off for Generalization of Neural Networks","Zitong Yang and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.11328",0,"","deception"],["Coherent Gradients: An Approach to Understanding Generalization in Gradient Descent-based Optimization","Satrajit Chatterjee","2020","paper","arXiv preprint","arxiv.org/abs/2002.10657",0,"",""],["Dividing the Ontology Alignment Task with Semantic Embeddings and Logic-based Modules","Ernesto Jiménez-Ruiz and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2003.05370",0,"","evals"],["How Low Should Fruit Hang Before We Pick It?","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LfGzAduBWzY5gq6FE/how-low-should-fruit-hang-before-we-pick-it",0,"",""],["New article from Oren Etzioni","Aryeh Englander","2020","blog","LessWrong","www.lesswrong.com/posts/phTQMkcH9Ttc4P9LB/new-article-from-oren-etzioni",0,"","forecasting"],["Other versions of \"No free lunch in value learning\"","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/LRYwpq8i9ym7Wuyoc/other-versions-of-no-free-lunch-in-value-learning",0,"",""],["Rewriting History with Inverse RL: Hindsight Inference for Policy Improvement","Benjamin Eysenbach and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.11089",0,"","policy"],["TanksWorld: A Multi-Agent Environment for AI Safety Research","Corban G. Rivera and 10 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.11174",0,"","agents"],["Subagents and impact measures, full and fully illustrated","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/mdQEraEZQLg7jtozn/subagents-and-impact-measures-full-and-fully-illustrated",0,"","agents policy"],["February 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/02/23/february-2020-newsletter/",0,"",""],["Neuron Shapley: Discovering the Responsible Neurons","Amirata Ghorbani and James Zou","2020","paper","arXiv preprint","arxiv.org/abs/2002.09815",0,"","interpretability"],["Attainable Utility Preservation: Empirical Results","TurnTrout and nealeratzlaff","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/4J4TA2ZF3wmSxhxuc/attainable-utility-preservation-empirical-results",0,"",""],["The Pragmatic Turn in Explainable Artificial Intelligence (XAI)","Andrés Páez","2020","paper","Minds and Machines, 29(3), 441-459, 2019","arxiv.org/abs/2002.09595",0,"","interpretability agents"],["Unsupervised Question Decomposition for Question Answering","Ethan Perez and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.09758",0,"",""],["Learning to Continually Learn","Shawn Beaulieu and 6 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.09571",0,"","deception robustness"],["Safe Imitation Learning via Fast Bayesian Reward Inference from Preferences","Daniel S. Brown and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.09089",0,"","reward-hacking evals policy"],["Will AI undergo discontinuous progress?","Sammy Martin","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/5WECpYABCT62TJrhY/will-ai-undergo-discontinuous-progress",0,"","forecasting"],["A Road Map to Strong Intelligence","Philip Paquette","2020","paper","arXiv preprint","arxiv.org/abs/2002.09044",0,"","ai-control deception agents"],["Curiosity Killed the Cat and the Asymptotically Optimal Agent","michaelcohen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fSC98Cy3zR9GsEPnT/curiosity-killed-the-cat-and-the-asymptotically-optimal",0,"","deception agents policy"],["Goal-directed = Model-based RL?","adamShimi","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Tux9WH4daKcxjEetQ/goal-directed-model-based-rl",0,"",""],["Tessellating Hills: a toy model for demons in imperfect search","DaemonicSigil","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/X7S3u5E4KktLp7gHz/tessellating-hills-a-toy-model-for-demons-in-imperfect",0,"",""],["The Problem with Metrics is a Fundamental Problem for AI","Rachel Thomas and David Uminsky","2020","paper","arXiv preprint","arxiv.org/abs/2002.08512",0,"","evals"],["[AN #87]: What might happen as deep learning scales even further?","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/69XPfonos795hD57o/an-87-what-might-happen-as-deep-learning-scales-even-further",0,"",""],["Estimating Training Data Influence by Tracing Gradient Descent","Garima Pruthi and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.08484",0,"","evals training-data"],["On unfixably unsafe AGI architectures","Steven Byrnes","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qvyv72fCiC46sxfPt/on-unfixably-unsafe-agi-architectures",0,"",""],["Counterfactuals versus the laws of physics","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/bqyCd38tACvKgqmXG/counterfactuals-versus-the-laws-of-physics",0,"",""],["Wireheading and discontinuity","Michele Campolo","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/KLNDgqQLfpFXbhQak/wireheading-and-discontinuity",0,"","reward-hacking"],["Appendix: mathematics of indexical impact measures","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/M9aoMixFLf8JFLRaP/appendix-mathematics-of-indexical-impact-measures",0,"",""],["Attainable Utility Preservation: Concepts","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/75oMAADr4265AGK3L/attainable-utility-preservation-concepts",0,"",""],["On the falsifiability of hypercomputation, part 2: finite input streams","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/PtaN3oMFPfAAuBNtw/on-the-falsifiability-of-hypercomputation-part-2-finite",0,"",""],["Does iterated amplification tackle the inner alignment problem?","JanBrauner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/RxutizkDNKzYCcNRv/does-iterated-amplification-tackle-the-inner-alignment",0,"","scalable-oversight"],["Reference Post: Trivial Decision Theory Problem","Chris_Leong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/XAeWHqQTWjJmzB4k6/reference-post-trivial-decision-theory-problem",0,"","theory"],["The Archimedean trap: Why traditional reinforcement learning will probably not yield AGI","Samuel Allen Alexander","2020","paper","Journal of Artificial General Intelligence 11(1): 70--85 (2020)","arxiv.org/abs/2002.10221",0,"","agents"],["Analyzing Differentiable Fuzzy Logic Operators","Emile van Krieken and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.06100",0,"",""],["Bayesian Evolving-to-Extinction","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/u9Azdu6Z7zFAhd4rK/bayesian-evolving-to-extinction",0,"",""],["Distinguishing definitions of takeoff","Matthew Barnett","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/YgNYA6pj2hPSDQiTE/distinguishing-definitions-of-takeoff",0,"","forecasting"],["The Catastrophic Convergence Conjecture","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/w6BtMqKRLxG9bNLMr/the-catastrophic-convergence-conjecture",0,"","instrumental-convergence"],["The Next Decade in AI: Four Steps Towards Robust Artificial Intelligence","Gary Marcus","2020","paper","arXiv preprint","arxiv.org/abs/2002.06177",0,"",""],["The Reasonable Effectiveness of Mathematics or: AI vs sandwiches","Vanessa Kosoy","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qpbYwTqKQG8G7mdFK/the-reasonable-effectiveness-of-mathematics-or-ai-vs",0,"",""],["A Simple Framework for Contrastive Learning of Visual Representations","Ting Chen and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.05709",0,"","deception robustness"],["CEB Improves Model Robustness","Ian Fischer and Alexander A. Alemi","2020","paper","arXiv preprint","arxiv.org/abs/2002.05380",0,"","benchmarks robustness"],["My personal cruxes for working on AI safety","Buck","2020","blog","EA Forum","forum.effectivealtruism.org/posts/Ayu5im98u8FeMWoBZ/my-personal-cruxes-for-working-on-ai-safety",0,"","forecasting"],["Our 2019 Fundraiser Review","Colm Ó Riain","2020","blog","intelligence.org","intelligence.org/2020/02/13/our-2019-fundraiser-review/",0,"",""],["The Conditional Entropy Bottleneck","Ian Fischer","2020","paper","arXiv preprint","arxiv.org/abs/2002.05379",0,"","evals robustness training-data"],["[AN #86]: Improving debate and factored cognition through human experiments","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/cZqPGDxbJcbShGwDn/an-86-improving-debate-and-factored-cognition-through-human",0,"",""],["A Bounded Measure for Estimating the Benefit of Visualization","Min Chen and 3 others","2020","paper","Entropy, 24(2), 228, 2022","arxiv.org/abs/2002.05282",0,"",""],["AI Impacts: Historic trends in technological progress","Aaron Gertler","2020","blog","EA Forum","forum.effectivealtruism.org/posts/APAD7PaEHgFyW3Nc4/ai-impacts-historic-trends-in-technological-progress",0,"",""],["AI safety: state of the field through quantitative lens","Mislav Juric and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.05671",0,"",""],["Demons in Imperfect Search","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/KnPN7ett8RszE79PH/demons-in-imperfect-search",0,"",""],["Growing Neural Cellular Automata","Alexander Mordvintsev and 2 others","2020","report","Distill","distill.pub/2020/growing-ca",0,"",""],["Leveraging Rationales to Improve Human Task Performance","Devleena Das and Sonia Chernova","2020","paper","arXiv preprint","arxiv.org/abs/2002.04202",0,"","interpretability evals"],["Short-Term AI Alignment as a Priority Cause","len.hoang.lnh","2020","blog","EA Forum","forum.effectivealtruism.org/posts/ptrY5McTdQfDy8o23/short-term-ai-alignment-as-a-priority-cause",0,"",""],["Attainable Utility Landscape: How The World Is Changed","TurnTrout","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/fj8eyc7QzqCaB8Wgm/attainable-utility-landscape-how-the-world-is-changed",0,"",""],["Gricean communication and meta-preferences","Charlie Steiner","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/8NpwfjFuEPMjTdriJ/gricean-communication-and-meta-preferences",0,"",""],["What can the principal-agent literature tell us about AI risk?","ac","2020","blog","EA Forum","forum.effectivealtruism.org/posts/N8pJdopFs7cLzAB6F/what-can-the-principal-agent-literature-tell-us-about-ai-1",0,"","agents"],["Effect of AlexNet on historic trends in image recognition","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/effect-of-alexnet-on-historic-trends-in-image-recognition/",0,"",""],["Historic trends in book production","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-book-production/",0,"",""],["Historic trends in bridge span length","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-bridge-span-length/",0,"",""],["Historic trends in chess AI","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-chess-ai/",0,"",""],["Historic trends in light intensity","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-light-intensity/",0,"",""],["Historic trends in long-range military payload delivery","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-long-range-military-payload-delivery/",0,"",""],["Historic trends in slow light technology","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-slow-light-technology/",0,"",""],["Historic trends in telecommunications performance","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-telecommunications-performance/",0,"",""],["Historic trends in the maximum superconducting temperature","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-the-maximum-superconducting-temperature/",0,"",""],["Historic trends in transatlantic message speed","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-transatlantic-message-speed/",0,"",""],["Incomplete case studies of discontinuous progress","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/incomplete-case-studies-of-discontinuous-progress/",0,"",""],["Penicillin and historic syphilis trends","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/penicillin-and-historic-syphilis-trends/",0,"",""],["What can the principal-agent literature tell us about AI risk?","Alexis Carlier","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Z5ZBPEgufmDsm7LAv/what-can-the-principal-agent-literature-tell-us-about-ai",0,"","agents"],["Effect of Eli Whitney’s cotton gin on historic trends in cotton ginning","Katja Grace","2020","blog","aiimpacts.org","aiimpacts.org/effect-of-eli-whitneys-cotton-gin-on-historic-trends-in-cotton-ginning/",0,"",""],["Exploring AI Futures Through Role Play","Shahar Avin and 2 others","2020","report","dl.acm.org","dl.acm.org/doi/10.1145/3375627.3375817",0,"",""],["Historic trends in flight airspeed records","Asya Bergal","2020","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-flight-airspeed-records/",0,"",""],["On the falsifiability of hypercomputation","jessicata","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/yjC5LmjSRD2hR9Pfa/on-the-falsifiability-of-hypercomputation",0,"",""],["Should Artificial Intelligence Governance be Centralised?: Design Lessons from History","Peter Cihon and 2 others","2020","report","dl.acm.org","dl.acm.org/doi/10.1145/3375627.3375857",0,"","governance"],["Student/Teacher Advising through Reward Augmentation","Cameron Reid","2020","paper","arXiv preprint","arxiv.org/abs/2002.02938",0,"","agents"],["The Windfall Clause: Distributing the Benefits of AI for the Common Good","Cullen O'Keefe and 5 others","2020","report","dl.acm.org","dl.acm.org/doi/10.1145/3375627.3375842",0,"","robustness"],["Plausibly, almost every powerful algorithm would be manipulative","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ez4zZQKWgC6fE3h9G/plausibly-almost-every-powerful-algorithm-would-be",0,"","deception instrumental-convergence"],["Quantifying Independently Reproducible Machine Learning","Edward Raff","2020","report","thegradient.pub","thegradient.pub/independently-reproducible-machine-learning/",0,"",""],["[AN #85]: The normative questions we should be asking for AI alignment, and a surprisingly good chatbot","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Mj259G5n5BxXXrZ7C/an-85-the-normative-questions-we-should-be-asking-for-ai",0,"","robustness"],["FHI Report: The Windfall Clause: Distributing the Benefits of AI for the Common Good","Cullen","2020","blog","EA Forum","forum.effectivealtruism.org/posts/iYCAoP3JgXxGAvMrr/fhi-report-the-windfall-clause-distributing-the-benefits-of",0,"","governance policy robustness"],["Synthesizing amplification and debate","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dJSD5RK6Qoidb3QY5/synthesizing-amplification-and-debate",0,"","scalable-oversight debate"],["Writeup: Progress on AI Safety via Debate","Beth Barnes and paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Br4xDbYu4Frwrb64a/writeup-progress-on-ai-safety-via-debate-1",0,"","scalable-oversight debate"],["Bridging the Gap: Providing Post-Hoc Symbolic Explanations for Sequential Decision-Making Problems with Inscrutable Representations","Sarath Sreedharan and 4 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.01080",0,"","evals benchmarks"],["Pessimism About Unknown Unknowns Inspires Conservatism","michaelcohen","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/RzAmPDNciirWKdtc7/pessimism-about-unknown-unknowns-inspires-conservatism",0,"","agents policy robustness monitoring"],["Philosophical self-ratification","jessicata","2020","blog","LessWrong","www.lesswrong.com/posts/g9dNMXKX2fqLgW9a9/philosophical-self-ratification",0,"","theory"],["What are the challenges and problems with programming law-breaking constraints into AGI?","MichaelStJules","2020","blog","EA Forum","forum.effectivealtruism.org/posts/qKXLpe7FNCdok3uvY/what-are-the-challenges-and-problems-with-programming-law",0,"","governance policy"],["Who owns artificial intelligence? A preliminary analysis of corporate intellectual property strategies and why they matter.","Nathan Calvin and Jade Leung","2020","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/GovAI-working-paper-Who-owns-AI-Apr2020.pdf",0,"",""],["Instrumental Occam?","abramdemski","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qqG2PdZ7pEcM6ev3S/instrumental-occam",0,"",""],["Preventing Imitation Learning with Adversarial Policy Ensembles","Albert Zhan and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2002.01059",0,"","policy"],["Brian Tse: Sino-Western cooperation in AI safety","EA Global","2020","blog","EA Forum","forum.effectivealtruism.org/posts/szwZkDBtW5sECHucy/brian-tse-sino-western-cooperation-in-ai-safety",0,"","governance policy"],["[AN #84] Reviewing AI alignment work in 2018-19","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6Rv9kLGmXrkqRrcK9/an-84-reviewing-ai-alignment-work-in-2018-19",0,"",""],["A high-precision abundance analysis of the nuclear benchmark star HD 20","Michael Hanke and 6 others","2020","paper","A&A 635, A104 (2020)","arxiv.org/abs/2001.11038",0,"","benchmarks"],["Slide deck: Introduction to AI Safety","Aryeh Englander","2020","blog","LessWrong","www.lesswrong.com/posts/8MXwoxNuicKtwmZm2/slide-deck-introduction-to-ai-safety",0,"",""],["Towards deconfusing values","Gordon Seidoh Worley","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/WAqG5BQMzAs34mpc2/towards-deconfusing-values",0,"",""],["Value uncertainty","MichaelA","2020","blog","LessWrong","www.lesswrong.com/posts/s6BGofzFbEr4Tmxkj/value-uncertainty",0,"",""],["AI Alignment 2018-19 Review","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/dKxX76SCfCvceJXHv/ai-alignment-2018-19-review",0,"","forecasting"],["AI Alignment 2018-2019 Review","Habryka","2020","blog","EA Forum","forum.effectivealtruism.org/posts/b2kSos3JqQCjKayHR/ai-alignment-2018-2019-review",0,"",""],["Algorithms vs Compute","johnswentworth","2020","blog","LessWrong","www.lesswrong.com/posts/QFuTYKhF4ouXTn9ML/algorithms-vs-compute",0,"","forecasting"],["Appendix: how a subagent could get powerful","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/sYjCeZTwA84pHkhBJ/appendix-how-a-subagent-could-get-powerful",0,"","agents"],["Towards Learning Multi-agent Negotiations via Self-Play","Yichuan Charlie Tang","2020","paper","arXiv preprint","arxiv.org/abs/2001.10208",0,"","evals agents policy"],["Using vector fields to visualise preferences and make them consistent","MichaelA and JustinShovelain","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ky988ePJvCRhmCwGo/using-vector-fields-to-visualise-preferences-and-make-them",0,"",""],["Towards a Human-like Open-Domain Chatbot","Daniel Adiwardana and 10 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.09977",0,"",""],["Silly rules improve the capacity of agents to learn stable enforcement and compliance behaviors","Raphael Köster and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.09318",0,"","agents"],["The two-layer model of human values, and problems with synthesizing preferences","Kaj_Sotala","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2yLn8iTrvHoEgqXcJ/the-two-layer-model-of-human-values-and-problems-with",0,"",""],["Formulating Reductive Agency in Causal Models","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/qrWFvMnRm4SkKnpRZ/formulating-reductive-agency-in-causal-models",0,"",""],["New paper: The Incentives that Shape Behaviour","RyanCarey","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/TgPCet7m9DnkuxyKP/new-paper-the-incentives-that-shape-behaviour",0,"",""],["Scaling Laws for Neural Language Models","Jared Kaplan and 9 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.08361",0,"","scaling-laws"],["What's a Good Prediction? Challenges in evaluating an agent's knowledge","Alex Kearney and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.08823",0,"","evals agents robustness"],["(A -> B) -> A in Causal DAGs","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/G25RBnBk5BNpv3KyF/a-greater-than-b-greater-than-a-in-causal-dags",0,"",""],["[AN #83]: Sample-efficient deep learning with ReMixMatch","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZrCsaCXrMTgrX9GzK/an-83-sample-efficient-deep-learning-with-remixmatch",0,"",""],["Concerns Surrounding CEV: A case for human friendliness first","ai-crotes","2020","blog","LessWrong","www.lesswrong.com/posts/haYD6N6BLvG7dkf25/concerns-surrounding-cev-a-case-for-human-friendliness-first",0,"",""],["Subjective Knowledge and Reasoning about Agents in Multi-Agent Systems","Shikha Singh and Deepak Khemani","2020","paper","arXiv preprint","arxiv.org/abs/2001.08016",0,"","agents"],["Designing for the Long Tail of Machine Learning","Martin Lindvall and Jesper Molin","2020","paper","arXiv preprint","arxiv.org/abs/2001.07455",0,"","training-data"],["Explaining Data-Driven Decisions made by AI Systems: The Counterfactual Approach","Carlos Fernández-Loría and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.07417",0,"",""],["How Doomed are Large Organizations?","Zvi","2020","blog","LessWrong","www.lesswrong.com/posts/ubHeLGc73iDvP6cTN/how-doomed-are-large-organizations",0,"","goodharts-law"],["Logical Representation of Causal Models","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/2DQHvGaH6C7dmwtdT/logical-representation-of-causal-models",0,"",""],["Inner alignment requires making assumptions about human values","Matthew Barnett","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6m5qqkeBTrqQsegGi/inner-alignment-requires-making-assumptions-about-human",0,"","agents"],["The Incentives that Shape Behaviour","Ryan Carey and 3 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.07118",0,"","agents"],["Gradient Surgery for Multi-Task Learning","Tianhe Yu and 5 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.06782",0,"","robustness"],["Teaching Software Engineering for AI-Enabled Systems","Christian Kästner and Eunsuk Kang","2020","paper","arXiv preprint","arxiv.org/abs/2001.06691",0,"",""],["[Link] EAF Research agenda: \"Cooperation, Conflict, and Transformative Artificial Intelligence\"","stefan.torges","2020","blog","EA Forum","forum.effectivealtruism.org/posts/Nn3gKtZptWua4PmtG/link-eaf-research-agenda-cooperation-conflict-and",0,"","governance policy theory"],["Activism by the AI Community: Analysing Recent Achievements and Future Prospects","Haydn Belfield","2020","paper","arXiv preprint","arxiv.org/abs/2001.06528",0,"",""],["Engineering AI Systems: A Research Agenda","Jan Bosch and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.07522",0,"",""],["Optimal by Design: Model-Driven Synthesis of Adaptation Strategies for Autonomous Systems","Yehia Elrakaiby and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.08525",0,"","evals agents"],["[AN #82]: How OpenAI Five distributed their training computation","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/6tikKda9LBzrkLfBJ/an-82-how-openai-five-distributed-their-training-computation",0,"",""],["ACDT: a hack-y acausal decision theory","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/9m2fzjNSJmd3yxxKG/acdt-a-hack-y-acausal-decision-theory",0,"","agents theory"],["In Defense of the Arms Races… that End Arms Races","Gentzel","2020","blog","LessWrong","www.lesswrong.com/posts/HpkZgmNskc2WwTy8N/in-defense-of-the-arms-races-that-end-arms-races",0,"","forecasting"],["January 2020 Newsletter","Rob Bensinger","2020","blog","intelligence.org","intelligence.org/2020/01/15/january-2020-newsletter/",0,"",""],["Predictors exist: CDT going bonkers... forever","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Kr76XzME7TFkN937z/predictors-exist-cdt-going-bonkers-forever",0,"","agents theory"],["Social and Governance Implications of Improved Data Efficiency","Aaron D. Tucker and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.05068",0,"","governance"],["What are the most pressing issues in short-term AI policy?","BrownHairedEevee","2020","blog","EA Forum","forum.effectivealtruism.org/posts/7hyLqG27skzfGR3ze/what-are-the-most-pressing-issues-in-short-term-ai-policy",0,"","governance policy"],["An Architectural Risk Analysis of Machine Learning Systems: Toward More Secure Machine Learning","Gary McGraw and 3 others","2020","report","garymcgraw.com","www.garymcgraw.com/wp-content/uploads/2020/02/BIML-ARA.pdf",0,"","agents"],["Artificial Intelligence, Values and Alignment","Iason Gabriel","2020","paper","Minds and Machines 2020","arxiv.org/abs/2001.09768",0,"",""],["Artificial Intelligence, Values and Alignment","Iason Gabriel","2020","blog","deepmind.com","www.deepmind.com/blog/artificial-intelligence-values-and-alignment",0,"",""],["Beyond Near- and Long-Term: Towards a Clearer Account of Research Priorities in AI Ethics and Society","Carina Prunkl and Jess Whittlestone","2020","paper","arXiv preprint","arxiv.org/abs/2001.04335",0,"",""],["Moral uncertainty: What kind of 'should' is involved?","MichaelA","2020","blog","LessWrong","www.lesswrong.com/posts/gvrojpfzizDvmPJJN/moral-uncertainty-what-kind-of-should-is-involved",0,"","theory"],["What is the relationship between Preference Learning and Value Learning?","Riccardo Volpato","2020","blog","LessWrong","www.lesswrong.com/posts/cPSqYG5qiRDpugrvq/what-is-the-relationship-between-preference-learning-and",0,"",""],["Malign generalization without internal search","Matthew Barnett","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/ynt9TD6PrYw6iT49m/malign-generalization-without-internal-search",0,"","agents robustness"],["Update on Ought's experiments on factored evaluation of arguments","Owain_Evans","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/pH3eKEAEupx8c2ep9/update-on-ought-s-experiments-on-factored-evaluation-of",0,"","evals"],["Evaluating Arguments One Step at a Time","Ought","2020","report","ought.org","ought.org/updates/2020-01-11-arguments",0,"","evals"],["I'm Cullen O'Keefe, a Policy Researcher at OpenAI, AMA","Cullen","2020","blog","EA Forum","forum.effectivealtruism.org/posts/9cx8TrLEooaw49cAr/i-m-cullen-o-keefe-a-policy-researcher-at-openai-ama",0,"","governance policy"],["Moral uncertainty vs related concepts","MichaelA","2020","blog","LessWrong","www.lesswrong.com/posts/oZsyK4SjnPe6HGia8/moral-uncertainty-vs-related-concepts",0,"","theory"],["Example: Markov Chain","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/KEZzAge6mgyo5GDi9/example-markov-chain",0,"",""],["Of arguments and wagers","paulfchristiano","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/aPsdGPCpcyPqkatgc/of-arguments-and-wagers",0,"",""],["Outer alignment and imitative amplification","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/33EKjmAdKFn3pbKPJ/outer-alignment-and-imitative-amplification",0,"","scalable-oversight goodharts-law"],["Visualizing the Impact of Feature Attribution Baselines","Pascal Sturmfels and 2 others","2020","report","Distill","distill.pub/2020/attribution-baselines",0,"",""],["2019-20 New Year review","Victoria Krakovna","2020","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2020/01/09/2019-20-new-year-review/",0,"",""],["Preference synthesis illustrated: Star Wars","Stuart_Armstrong","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/Nfizy2uRNkZmX3AYB/preference-synthesis-illustrated-star-wars",0,"",""],["The Logic of Strategic Assets: From Oil to Artificial Intelligence","Jeffrey Ding and Allan Dafoe","2020","paper","arXiv preprint","arxiv.org/abs/2001.03246",0,"","policy"],["(Double-)Inverse Embedded Agency Problem","shminux","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/itGmH2AknmjWyAwj8/double-inverse-embedded-agency-problem",0,"","theory"],["[AN #81]: Universality as a potential solution to conceptual difficulties in intent alignment","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/3kzFPA5uuaGZWg4PS/an-81-universality-as-a-potential-solution-to-conceptual",0,"",""],["Algorithmic Fairness from a Non-ideal Perspective","Sina Fazelpour and Zachary C. Lipton","2020","paper","arXiv preprint","arxiv.org/abs/2001.09773",0,"",""],["How to Throw Away Information in Causal DAGs","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/zFGGHGfhYsGNnh7Kp/how-to-throw-away-information-in-causal-dags",0,"",""],["Definitions of Causal Abstraction: Reviewing Beckers & Halpern","johnswentworth","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/gJ76SLJAaKZrFCRTj/definitions-of-causal-abstraction-reviewing-beckers-and",0,"",""],["Morality vs related concepts","MichaelA","2020","blog","LessWrong","www.lesswrong.com/posts/dXT5G9xEAddac8H2J/morality-vs-related-concepts",0,"","theory"],["Exploring safe exploration","evhub","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/NBffcjqm2P4dNbjrE/exploring-safe-exploration",0,"",""],["Dissolving Confusion around Functional Decision Theory","scasper","2020","blog","LessWrong","www.lesswrong.com/posts/xoQRz8tBvsznMXTkt/dissolving-confusion-around-functional-decision-theory",0,"","theory"],["Auditing and Debugging Deep Learning Models via Decision Boundaries: Individual-level and Group-level Analysis","Roozbeh Yousefzadeh and Dianne P. O'Leary","2020","paper","arXiv preprint","arxiv.org/abs/2001.00682",0,"",""],["[AN #80]: Why AI risk might be solved without additional intervention from longtermists","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/QknPz9JQTQpGdaWDp/an-80-why-ai-risk-might-be-solved-without-additional",0,"",""],["Making decisions when both morally and empirically uncertain","MichaelA","2020","blog","LessWrong","www.lesswrong.com/posts/eYiDjCNJrR3w3WcMM/making-decisions-when-both-morally-and-empirically-uncertain",0,"","theory"],["[AN #79]: Recursive reward modeling as an alignment technique integrated with deep RL","Rohin Shah","2020","blog","AI Alignment Forum","www.alignmentforum.org/posts/EoY6P6mpz7ZozhAxm/an-79-recursive-reward-modeling-as-an-alignment-technique",0,"","scalable-oversight"],["A Framework for Democratizing AI","Shakkeel Ahmed and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2001.00818",0,"","policy"],["AGI Safety From First Principles","Richard Ngo","2020","report","drive.google.com","drive.google.com/file/d/1uK7NhdSKprQKZnRjU58X7NLA1auXlWHt/view?pli=1",0,"",""],["AI Paradigms and AI Safety: Mapping Artefacts and Techniques to Safety Issues","Jose Hernandez-Orallo and 4 others","2020","report","ecai2020.eu","ecai2020.eu/papers/1364_paper.pdf",0,"",""],["Automating reasoning about the future at Ought","Jungwon Byun and Andreas Stuhlmüller","2020","report","ought.org","ought.org/updates/2020-11-09-forecasting",0,"",""],["Avoiding Negative Side Effects due to Incomplete Knowledge of AI Systems","Sandhya Saisubramanian and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2008.12146",0,"","agents"],["Avoiding Side Effects in Complex Environments","Alexander Matt Turner* and 2 others","2020","paper","arXiv preprint","arxiv.org/abs/2006.06547",0,"","benchmarks"],["Building Trust Through Testing","Michèle A Flournoy and 2 others","2020","report","cset.georgetown.edu","cset.georgetown.edu/wp-content/uploads/Building-Trust-Through-Testing.pdf",0,"","interpretability evals red-teaming robustness"],["Canaries in Technology Mines: Warning Signs of Transformative Progress in AI","Carla Zoe Cremer and Jess Whittlestone","2020","report","dmip.webs.upv.es","dmip.webs.upv.es/EPAI2020/papers/EPAI_2020_paper_4.pdf",0,"",""],["Cheating Death in Damascus","Benjamin A. Levinstein and 2 others","2020","report","pdcnet.org","www.pdcnet.org/oom/service?url_ver=Z39.88-2004&rft_val_fmt=&rft.imuse_id=jphil_2020_0117_0005_0237_0266&svc_id=info:www.pdcnet.org/collection",0,"",""],["Choice Set Misspeciﬁcation in Reward Inference","Rachel Freedman and 2 others","2020","report","ceur-ws.org","ceur-ws.org/Vol-2640/paper_14.pdf",0,"",""],["Classification of global catastrophic risks connected with artificial intelligence","Alexey Turchin and David Denkenberger","2020","report","link.springer.com","link.springer.com/epdf/10.1007/s00146-018-0845-5",0,"",""],["Decision Points in AI Governance","Jessica Cussins Newman","2020","report","cltc.berkeley.edu","cltc.berkeley.edu/wp-content/uploads/2020/05/Decision_Points_AI_Governance.pdf",0,"","governance policy"],["Defence in Depth Against Human Extinction: Prevention, Response, Resilience, and Why They All Matter","Owen Cotton‐Barratt and 2 others","2020","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/abs/10.1111/1758-5899.12786",0,"",""],["Fragmentation and the Future: Investigating Architectures for International AI Governance","Peter Cihon and 2 others","2020","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/abs/10.1111/1758-5899.12890",0,"","governance"],["From the Standard Model of AI to Provably Beneficial Systems","Stuart Russell and Caroline Jeanmaire","2020","report","n.sinaimg.cn","n.sinaimg.cn/tech/f34884a9/20200501/GlobalAIGovernancein2019.pdf",0,"",""],["International evaluation of an AI system for breast cancer screening","Scott Mayer McKinney * and 30 others","2020","blog","deepmind.com","www.deepmind.com/blog/international-evaluation-of-an-ai-system-for-breast-cancer-screening",0,"","evals"],["Learning to summarize with human feedback","Nisan Stiennon and 8 others","2020","report","proceedings.neurips.cc","proceedings.neurips.cc/paper_files/paper/2020/file/1f89885d556929e98d3ef9b86448f951-Paper.pdf",0,"","rlhf"],["Open Problems in Cooperative AI","Allan Dafoe and 7 others","2020","paper","arXiv preprint","arxiv.org/abs/2012.08630",0,"",""],["OpenAI Microscope","OpenAI","2020","report","microscope.openai.com","microscope.openai.com/",0,"",""],["Pragmatic-Pedagogic Value Alignment","Jaime F. Fisac and 9 others","2020","report","link.springer.com","link.springer.com/10.1007/978-3-030-28619-4_7",0,"",""],["Responsive safety in reinforcement learning by pid lagrangian methods","Adam Stooke and 2 others","2020","report","proceedings.mlr.press","proceedings.mlr.press/v119/stooke20a/stooke20a.pdf",0,"",""],["Safer ML paradigms team: the story – AI Safety Research Program","AI Safety Camp","2020","report","aisrp.org","aisrp.org/?page_id=169",0,"",""],["Since figuring out human values is hard, what about, say, monkey values?","shminux","2020","blog","LessWrong","www.lesswrong.com/posts/2kphKANEE2NA9JKoN/since-figuring-out-human-values-is-hard-what-about-say",0,"",""],["The MAGICAL Benchmark for Robust Imitation","Sam Toyer and 3 others","2020","report","papers.nips.cc","papers.nips.cc/paper/2020/hash/d464b5ac99e74462f321c06ccacc4bff-Abstract.html",0,"","benchmarks"],["The Question of Comparative Advantage in Artificial Intelligence: Enduring Strengths and Emerging Challenges for the United States","Andrew Imbrie and 2 others","2020","report","cset.georgetown.edu","cset.georgetown.edu/research/the-question-of-comparative-advantage-in-artificial-intelligence-enduring-strengths-and-emerging-challenges-for-the-united-states/",0,"",""],["Towards Cooperation in Learning Games","Jesse Clifton and Maxime Riché","2020","report","longtermrisk.org","longtermrisk.org/files/toward_cooperation_learning_games_oct_2020.pdf",0,"",""],["human psycholinguists: a critical appraisal","nostalgebraist","2019","blog","LessWrong","www.lesswrong.com/posts/ZFtesgbY9XwtqqyZ5/human-psycholinguists-a-critical-appraisal",0,"","forecasting"],["KOLSITAN, a tiny video game","Tamsin Leake","2019","blog","carado.moe","carado.moe/kolsitan.html",0,"",""],["Reward-Conditioned Policies","Aviral Kumar and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.13465",0,"","benchmarks policy robustness"],["Uncertainty-Based Out-of-Distribution Classification in Deep Reinforcement Learning","Andreas Sedlmeier and 4 others","2019","paper","Proceedings of the 12th International Conference on Agents and\n  Artificial Intelligence - Volume 2: ICAART, 2020, ISBN 978-989-758-395-7,\n  pages 522-529","arxiv.org/abs/2001.00496",0,"","evals agents robustness training-data"],["Making decisions under moral uncertainty","MichaelA","2019","blog","LessWrong","www.lesswrong.com/posts/dX7vNKg4vex5vxWCW/making-decisions-under-moral-uncertainty-1",0,"","theory"],["Asking the Right Questions: Learning Interpretable Action Models Through Query Answering","Pulkit Verma and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.12613",0,"","interpretability evals agents policy"],["Safe exploration and corrigibility","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/87Y7w73phjBxnPyPD/safe-exploration-and-corrigibility",0,"",""],["Conversation on AI risk with Adam Gleave","AI Impacts","2019","blog","EA Forum","forum.effectivealtruism.org/posts/SiF3iWGSFn562vbGr/conversation-on-ai-risk-with-adam-gleave",0,"",""],["Critiquing \"What failure looks like\"","Grue_Slinky","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Q8Z8yoG4tBaowBHwk/critiquing-what-failure-looks-like",0,"",""],["The Offense-Defense Balance of Scientific Knowledge: Does Publishing AI Research Reduce Misuse?","Toby Shevlane and Allan Dafoe","2019","paper","arXiv preprint","arxiv.org/abs/2001.00463",0,"","robustness"],["[AN #78] Formalizing power and instrumental convergence, and the end-of-year AI safety charity comparison","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/D7CY29s2D6HJirqcF/an-78-formalizing-power-and-instrumental-convergence-and-the",0,"","instrumental-convergence"],["Brief summary of key disagreements in AI Risk","Aryeh Englander","2019","blog","EA Forum","forum.effectivealtruism.org/posts/HayWBGerpYFk3GsZR/brief-summary-of-key-disagreements-in-ai-risk",0,"",""],["New paper: (When) is Truth-telling Favored in AI debate?","VojtaKovarik","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/RQoSCs9SePDMLJvfz/new-paper-when-is-truth-telling-favored-in-ai-debate",0,"",""],["Another AI Winter?","PeterMcCluskey","2019","blog","LessWrong","www.lesswrong.com/posts/zihjMujStyktz64ie/another-ai-winter",0,"","forecasting"],["Comparison of naturally evolved and engineered solutions","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/comparison-of-naturally-evolved-and-engineered-solutions/",0,"",""],["Walsh 2017 survey","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/walsh-2017-survey/",0,"",""],["Conversation with Adam Gleave","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/conversation-with-adam-gleave/",0,"","robustness"],["Defining AI in Policy versus Practice","P. M. Krafft and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.11095",0,"","policy"],["Effects of breech loading rifles on historic trends in firearm progress","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/effects-of-breech-loading-rifles-on-historic-trends-in-firearm-progress/",0,"",""],["Historic trends in ship size","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-ship-size/",0,"",""],["Humans Are Embedded Agents Too","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WJzsTmsDctYCCyMfy/humans-are-embedded-agents-too",0,"","agents theory"],["Might humans not be the most intelligent animals?","Matthew Barnett","2019","blog","LessWrong","www.lesswrong.com/posts/XjuT9vgBfwXPxsdfN/might-humans-not-be-the-most-intelligent-animals",0,"","forecasting"],["Section 7: Foundations of Rational Agency","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/sMhJsRfLXAg87EEqT/section-7-foundations-of-rational-agency",0,"","theory"],["Questions to Guide the Future of Artificial Intelligence Research","Jordan Ott","2019","paper","arXiv preprint","arxiv.org/abs/1912.10305",0,"",""],["The Counterfactual Prisoner's Dilemma","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/sY2rHNcWdg94RiSSR/the-counterfactual-prisoner-s-dilemma",0,"",""],["Clarifying Power-Seeking and Instrumental Convergence","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cwpKagyTvqSyAJB7q/clarifying-power-seeking-and-instrumental-convergence",0,"","instrumental-convergence power-seeking"],["Mastering Complex Control in MOBA Games with Deep Reinforcement Learning","Deheng Ye and 17 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.09729",0,"","agents"],["Retrospective on the specification gaming examples list","Victoria Krakovna","2019","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2019/12/20/retrospective-on-the-specification-gaming-examples-list/",0,"","specification-gaming"],["Sections 5 & 6: Contemporary Architectures, Humans in the Loop","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4GuKi9wKYnthr8QP9/sections-5-and-6-contemporary-architectures-humans-in-the",0,"",""],["2019 AI Alignment Literature Review and Charity Comparison","Larks","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SmDziGM9hBjW9DKmf/2019-ai-alignment-literature-review-and-charity-comparison",0,"","governance"],["Causal Abstraction Intro","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yD9GLtQgp8vAfndL8/causal-abstraction-intro",0,"",""],["When Goodharting is optimal: linear vs diminishing returns, unlikely vs likely, and other factors","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/megKzKKsoecdYqwb7/when-goodharting-is-optimal-linear-vs-diminishing-returns",0,"","goodharts-law"],["[AN #77]: Double descent: a unification of statistical theory and modern ML practice","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/LYdvzXF6E4iXM2ZSD/an-77-double-descent-a-unification-of-statistical-theory-and",0,"",""],["Abstraction, Causality, and Embedded Maps: Here Be Monsters","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ipCAL4tx7jcJsFasY/abstraction-causality-and-embedded-maps-here-be-monsters",0,"",""],["Inductive biases stick around","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/nGqzNC6uNueum2w8T/inductive-biases-stick-around",0,"",""],["Why we need an AI-resilient society","Thomas Bartz-Beielstein","2019","paper","arXiv preprint","arxiv.org/abs/1912.08786",0,"",""],["A dilemma for prosaic AI alignment","Daniel Kokotajlo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jYdAxH8BarPT4fqnb/a-dilemma-for-prosaic-ai-alignment",0,"","scalable-oversight debate"],["Counterfactual Induction","Diffractor","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/EAqHkKtbefvyRs4nw/counterfactual-induction",0,"",""],["Counterfactual Induction (Algorithm Sketch, Fixpoint proof)","Diffractor","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/xBoBmPtgvwdfqm2r5/counterfactual-induction-algorithm-sketch-fixpoint-proof",0,"",""],["Counterfactual Induction (Lemma 4)","Diffractor","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cu4v9MHGuhLnDQTuF/counterfactual-induction-lemma-4",0,"",""],["Counterfactual Mugging: Why should you pay?","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/h9qQQA3g8dwq6RRTo/counterfactual-mugging-why-should-you-pay",0,"","theory"],["Generative Teaching Networks: Accelerating Neural Architecture Search by Learning to Generate Synthetic Training Data","Felipe Petroski Such and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.07768",0,"","evals agents training-data"],["Is Causality in the Map or the Territory?","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZBYE2F5DBiZtj6m95/is-causality-in-the-map-or-the-territory",0,"",""],["Sections 1 & 2: Introduction, Strategy and Governance","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/KMocAf9jnAKc2jXri/sections-1-and-2-introduction-strategy-and-governance",0,"","governance"],["Sections 3 & 4: Credibility, Peaceful Bargaining Mechanisms","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/8xKhCbNrdP4gaA8c3/sections-3-and-4-credibility-peaceful-bargaining-mechanisms",0,"",""],["More Data Can Hurt for Linear Regression: Sample-wise Double Descent","Preetum Nakkiran","2019","paper","arXiv preprint","arxiv.org/abs/1912.07242",0,"","training-data"],["Should Artificial Intelligence Governance be Centralised? Six Design Lessons from History","Peter Cihon and 2 others","2019","report","cser.ac.uk","www.cser.ac.uk/media/uploads/files/Cihon_et_al-_2019-_Should_AI_Governance_be_Centralised.pdf",0,"","governance robustness"],["Acknowledgements & References","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/XKWGgyCyGhkm73fhm/acknowledgements-and-references",0,"",""],["Is the term mesa optimizer too narrow?","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/nFDXq7HTv9Xugcqaw/is-the-term-mesa-optimizer-too-narrow",0,"","policy robustness"],["But exactly how complex and fragile?","Katja_Grace","2019","blog","EA Forum","forum.effectivealtruism.org/posts/fRY74NeM3cxCdNPth/but-exactly-how-complex-and-fragile",0,"",""],["Dota 2 with Large Scale Deep Reinforcement Learning","OpenAI and 26 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.06680",0,"",""],["Preface to CLR's Research Agenda on Cooperation, Conflict, and TAI","JesseClifton","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/DbuCdEbkh4wL5cjJ5/preface-to-clr-s-research-agenda-on-cooperation-conflict-and",0,"",""],["Examples of Causal Abstraction","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Expvyb6nndbjqigRL/examples-of-causal-abstraction",0,"",""],["Causal Abstraction Toy Model: Medical Sensor","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/S8WZ2rav9BqFAZoRM/causal-abstraction-toy-model-medical-sensor",0,"",""],["Linear Mode Connectivity and the Lottery Ticket Hypothesis","Jonathan Frankle and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05671",0,"",""],["Regulatory Markets for AI Safety","Jack Clark and Gillian K. Hadfield","2019","paper","arXiv preprint","arxiv.org/abs/2001.00078",0,"","governance"],["What Can Learned Intrinsic Rewards Capture?","Zeyu Zheng and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05500",0,"","agents policy robustness"],["Deep Bayesian Reward Learning from Preferences","Daniel S. Brown and Scott Niekum","2019","paper","arXiv preprint","arxiv.org/abs/1912.04472",0,"","reward-hacking evals agents policy"],["Predictive coding = RL + SL + Bayes + MPC","Steven Byrnes","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cfvBm2kBtFTgxBB7s/predictive-coding-rl-sl-bayes-mpc",0,"",""],["Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","Akanksha Atrey and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05743",0,"","agents"],["Meta-Learning without Memorization","Mingzhang Yin and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.03820",0,"","training-data"],["Counterfactuals: Smoking Lesion vs. Newcomb's","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/8Hr95c37nadXCxzh7/counterfactuals-smoking-lesion-vs-newcomb-s",0,"",""],["The Lesson To Unlearn","Ben Pace","2019","blog","LessWrong","www.lesswrong.com/posts/soj9YdzDCWaB8uSTP/the-lesson-to-unlearn",0,"","goodharts-law"],["Value-of-Information based Arbitration between Model-based and Model-free Control","Krishn Bera and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05453",0,"","evals"],["Are Humans 'Human Compatible'?","Matt Boyd","2019","blog","EA Forum","forum.effectivealtruism.org/posts/voEDjdnZyWkxi54SR/are-humans-human-compatible",0,"",""],["Comment on Coherence arguments do not imply goal directed behavior","Ronny Fernandez","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/EnN7cm3KaRrEAuWfa/comment-on-coherence-arguments-do-not-imply-goal-directed",0,"","theory"],["Understanding “Deep Double Descent”","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/FRv7ryoqtvSuqBxuT/understanding-deep-double-descent",0,"",""],["What is Abstraction?","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wuJpYLcMEBz4kcgAn/what-is-abstraction-1",0,"",""],["December 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/12/05/december-2019-newsletter/",0,"",""],["Deep Ensembles: A Loss Landscape Perspective","Stanislav Fort and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.02757",0,"","evals robustness"],["Historic trends in transatlantic passenger travel","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-transatlantic-passenger-travel/",0,"",""],["Learning Human Objectives by Evaluating Hypothetical Behavior","Siddharth Reddy and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05652",0,"","reward-hacking evals agents policy"],["Oracles: reject all deals - break superrationality, with superrationality","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/6XCTppoPAMdKCPFb4/oracles-reject-all-deals-break-superrationality-with-1",0,"",""],["Oracles: reject all deals - break superrationality, with superrationality","Stuart_Armstrong","2019","blog","LessWrong","www.lesswrong.com/posts/6XCTppoPAMdKCPFb4/oracles-reject-all-deals-break-superrationality-with-1",0,"",""],["Seeking Power is Often Convergently Instrumental in MDPs","TurnTrout and Logan Riggs","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/6DuJxY8X45Sco4bS2/seeking-power-is-often-convergently-instrumental-in-mdps",0,"","instrumental-convergence"],["Values, Valence, and Alignment","Gordon Seidoh Worley","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ALvnz3DrjHwmLG29F/values-valence-and-alignment",0,"",""],["What are some non-purely-sampling ways to do deep RL?","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ca3sCRGfWvXvYC5YC/what-are-some-non-purely-sampling-ways-to-do-deep-rl",0,"","agents"],["\"Fully\" acausal trade","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/MHHzLfAQBZzieGBjq/fully-acausal-trade",0,"",""],["[AN #76]: How dataset size affects robustness, and benchmarking safe exploration by measuring constraint violations","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SXoHj7DTAjAsfJrcs/an-76-how-dataset-size-affects-robustness-and-benchmarking",0,"","benchmarks robustness"],["Learning Efficient Representation for Intrinsic Motivation","Ruihan Zhao and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.02624",0,"","agents"],["Recent Progress in the Theory of Neural Networks","interstice","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/KrQvZM8uFjSTJ7hq3/recent-progress-in-the-theory-of-neural-networks-1",0,"",""],["Adaptive Online Planning for Continual Lifelong Learning","Kevin Lu and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.01188",0,"","deception policy"],["Dream to Control: Learning Behaviors by Latent Imagination","Danijar Hafner and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.01603",0,"","agents"],["Measuring the intelligence of an idealized mechanical knowing agent","Samuel Allen Alexander","2019","paper","arXiv preprint","arxiv.org/abs/1912.09571",0,"","agents"],["SafeLife 1.0: Exploring Side Effects in Complex Environments","Carroll L. Wainwright and Peter Eckersley","2019","paper","CEUR Workshop Proceedings, 2560 (2020) 117-127","arxiv.org/abs/1912.01217",0,"","benchmarks agents policy"],["A list of good heuristics that the case for AI x-risk fails","David Scott Krueger (formerly: capybaralet)","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/bd2K3Jdz82csjCFob/a-list-of-good-heuristics-that-the-case-for-ai-x-risk-fails",0,"","evals robustness"],["Deep Learning for Symbolic Mathematics","Guillaume Lample and François Charton","2019","paper","arXiv preprint","arxiv.org/abs/1912.01412",0,"","training-data"],["MIRI’s 2019 Fundraiser","Malo Bourgon","2019","blog","intelligence.org","intelligence.org/2019/12/02/miris-2019-fundraiser/",0,"",""],["What I talk about when I talk about AI x-risk: 3 core claims I want machine learning researchers to address.","David Scott Krueger (formerly: capybaralet)","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/bJdaB2Mz4mBvwFBeb/what-i-talk-about-when-i-talk-about-ai-x-risk-3-core-claims-1",0,"","goodharts-law instrumental-convergence"],["A Parametric, Resource-Bounded Generalization Of Löb’s Theorem, And A Robust Cooperation Criterion For Open-Source Game Theory","Andrew Critch","2019","report","cambridge.org","www.cambridge.org/core/product/identifier/S0022481217000421/type/journal_article",0,"",""],["Algorithmic Decision-Making and the Control Problem","John Zerilli and 3 others","2019","report","link.springer.com","link.springer.com/10.1007/s11023-019-09513-7",0,"",""],["CHAI Newsletter #3 2019","CHAI","2019","report","drive.google.com","drive.google.com/file/d/1BxMJaWmF39r0b3DH40PiPzk5oEzCD3GH/view?usp=sharing",0,"",""],["CHAI Newsletter #4 2019","CHAI","2019","report","drive.google.com","drive.google.com/file/d/1YNNYmJDoHMl7H7bRtwp8jeY_SFlmFL0i/view?usp=sharing",0,"",""],["Interactive AI with a Theory of Mind","Mustafa Mert Çelikok and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05284",0,"","agents"],["Neural Networks and Deep Learning, Chapters 1-6 (more entry-level)","Michael Nielsen","2019","report","neuralnetworksanddeeplearning.com","neuralnetworksanddeeplearning.com/",0,"",""],["Counterfactuals as a matter of Social Convention","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9rtWTHsPAf2mLKizi/counterfactuals-as-a-matter-of-social-convention",0,"",""],["Useful Does Not Mean Secure","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/mdau2DBSMi5bWXPGA/useful-does-not-mean-secure",0,"","interpretability"],["What's been written about the nature of \"son-of-CDT\"?","Liam Donovan","2019","blog","LessWrong","www.lesswrong.com/posts/si76HRBRvewsRMeWP/what-s-been-written-about-the-nature-of-son-of-cdt",0,"","theory"],["Induction of Subgoal Automata for Reinforcement Learning","Daniel Furelos-Blanco and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.13152",0,"","evals agents"],["Anti-Alignments -- Measuring The Precision of Process Models and Event Logs","Thomas Chatain and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.05907",0,"","deception"],["Giving Tuesday 2019","Colm Ó Riain","2019","blog","intelligence.org","intelligence.org/2019/11/28/giving-tuesday-2019/",0,"",""],["[AN #75]: Solving Atari and Go with learned game models, and thoughts from a MIRI employee","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NSCBF7MTLF2HdhEnD/an-75-solving-atari-and-go-with-learned-game-models-and",0,"",""],["The relationship between trust in AI and trustworthy machine learning technologies","Ehsan Toreini and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1912.00782",0,"","policy"],["The Transformative Potential of Artificial Intelligence","Ross Gruetzemacher and Jess Whittlestone","2019","paper","arXiv preprint","arxiv.org/abs/1912.00747",0,"",""],["A test for symbol grounding methods: true zero-sum games","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/JpEPKbXiTvmyqYdTr/a-test-for-symbol-grounding-methods-true-zero-sum-games-1",0,"",""],["Thoughts on implementing corrigible robust alignment","Steven Byrnes","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/8W5gNgEKnyAscg8BF/thoughts-on-implementing-corrigible-robust-alignment",0,"",""],["Breaking Oracles: superrationality and acausal trade","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/42z4k8Co5BuHMBvER/breaking-oracles-superrationality-and-acausal-trade",0,"",""],["November 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/11/25/november-2019-newsletter/",0,"",""],["Scaling Out-of-Distribution Detection for Real-World Settings","Dan Hendrycks and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.11132",0,"","evals benchmarks"],["Thoughts on Robin Hanson's AI Impacts interview","Steven Byrnes","2019","blog","LessWrong","www.lesswrong.com/posts/w6AzbZR7ZQxWuAwKR/thoughts-on-robin-hanson-s-ai-impacts-interview",0,"","forecasting"],["Analysing: Dangerous messages from future UFAI via Oracles","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/6WbLRLdmTL4JxxvCq/analysing-dangerous-messages-from-future-ufai-via-oracles",0,"",""],["Ultra-simplified research agenda","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/m2bwD87ctjJDXC3SZ/ultra-simplified-research-agenda",0,"",""],["A Brief Intro to Domain Theory","Diffractor","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4C4jha5SdReWgg7dF/a-brief-intro-to-domain-theory",0,"",""],["Defining AI wireheading","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/vXzM5L6njDZSf4Ftk/defining-ai-wireheading",0,"","reward-hacking specification-gaming"],["ReMixMatch: Semi-Supervised Learning with Distribution Alignment and Augmentation Anchoring","David Berthelot and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.09785",0,"","training-data"],["[AN #74]: Separating beneficial AI into competence, alignment, and coping with impacts","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/X2fRsTjd2kQ89pipE/an-74-separating-beneficial-ai-into-competence-alignment-and",0,"",""],["AI safety scholarships look worth-funding (if other funding is sane)","anon-a","2019","blog","EA Forum","forum.effectivealtruism.org/posts/fbw7mg2CzBiHqRibr/ai-safety-scholarships-look-worth-funding-if-other-funding-2",0,"",""],["Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model","Julian Schrittwieser and 11 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.08265",0,"","interpretability policy"],["Planning with Goal-Conditioned Policies","Soroush Nasiriany and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.08453",0,"","evals"],["Impossible moral problems and moral authority","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pW6YJEzoRFe9cshuN/impossible-moral-problems-and-moral-authority",0,"",""],["Self-Fulfilling Prophecies Aren't Always About Self-Awareness","John_Maxwell","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yArZKCEheZt8GkK6p/self-fulfilling-prophecies-aren-t-always-about-self",0,"","situational-awareness"],["The Goodhart Game","John_Maxwell","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WnPEe99YuyRxktMD3/the-goodhart-game",0,"","goodharts-law robustness"],["The new dot com bubble is here: it’s called online advertising","Gordon Seidoh Worley","2019","blog","LessWrong","www.lesswrong.com/posts/icPvmaB4fBxy7Divt/the-new-dot-com-bubble-is-here-it-s-called-online",0,"","goodharts-law"],["The Value Definition Problem","Sammy Martin","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/W95gbuognJu5WxkTW/the-value-definition-problem",0,"","robustness"],["How common is it for one entity to have a 3+ year technological lead on its nearest competitor?","Daniel Kokotajlo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXikQ87FFw3oPPaYh/how-common-is-it-for-one-entity-to-have-a-3-year",0,"","forecasting"],["I'm Buck Shlegeris, I do research and outreach at MIRI, AMA","Buck","2019","blog","EA Forum","forum.effectivealtruism.org/posts/tDk57GhrdK54TWzPY/i-m-buck-shlegeris-i-do-research-and-outreach-at-miri-ama",0,"",""],["Evolution of Modularity","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/JBFHzfPkXHB2XfDGj/evolution-of-modularity",0,"",""],["[AN #73]: Detecting catastrophic failures by learning how agents tend to break","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/mQqFNbvD5mYrCQPKE/an-73-detecting-catastrophic-failures-by-learning-how-agents",0,"","agents"],["Conversation with Robin Hanson","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/conversation-with-robin-hanson/",0,"","agents forecasting"],["Momentum Contrast for Unsupervised Visual Representation Learning","Kaiming He and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.05722",0,"",""],["On AI Weapons","kbog","2019","blog","EA Forum","forum.effectivealtruism.org/posts/vdqBn65Qaw77MpqXz/on-ai-weapons",0,"",""],["Robin Hanson on the futurist focus on AI","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/robin-hanson-on-the-futurist-focus-on-ai/",0,"",""],["A conversation with Rohin Shah","AI Impacts","2019","blog","EA Forum","forum.effectivealtruism.org/posts/Y2hkJ5STZfBCyRG9r/a-conversation-with-rohin-shah",0,"","forecasting"],["What I’ll be doing at MIRI","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ptmmK9PWgYTuWToaZ/what-i-ll-be-doing-at-miri",0,"",""],["(When) Is Truth-telling Favored in AI Debate?","Vojtěch Kovařík and Ryan Carey","2019","paper","arXiv preprint","arxiv.org/abs/1911.04266",0,"",""],["AI policy careers in the EU","Lauro Langosco","2019","blog","EA Forum","forum.effectivealtruism.org/posts/XGPW25NZHq2WHbK9w/ai-policy-careers-in-the-eu",0,"","governance policy"],["Operationalizing Newcomb's Problem","ErickBall","2019","blog","LessWrong","www.lesswrong.com/posts/X6f3KGYgnXxCHAnAq/operationalizing-newcomb-s-problem",0,"","theory"],["Self-training with Noisy Student improves ImageNet classification","Qizhe Xie and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.04252",0,"",""],["The Credit Assignment Problem","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ajcq9xWi2fmgn8RBJ/the-credit-assignment-problem",0,"",""],["AISC3: Research Summaries","Kristina Němcová","2019","blog","aisafety.camp","aisafety.camp/2019/11/07/aisc3-research-summaries/",0,"",""],["Uber Self-Driving Crash","jefftk","2019","blog","LessWrong","www.lesswrong.com/posts/tTg4bn5rxHYqQJXhD/uber-self-driving-crash",0,"","forecasting"],["[AN #72]: Alignment, robustness, methodology, and system building as research priorities for AI safety","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hPJMum5CNH5MKe27C/an-72-alignment-robustness-methodology-and-system-building",0,"","robustness"],["A mechanistic model of meditation","Kaj_Sotala","2019","blog","LessWrong","www.lesswrong.com/posts/WYmmC3W6ZNhEgAmWG/a-mechanistic-model-of-meditation",0,"","agents theory"],["AI Alignment Research Overview (by Jacob Steinhardt)","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/7GEviErBXcjJsbSeD/ai-alignment-research-overview-by-jacob-steinhardt",0,"",""],["Etzioni 2016 survey","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/etzioni-2016-survey/",0,"",""],["Nonverbal Robot Feedback for Human Teachers","Sandy H. Huang and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.02320",0,"","interpretability"],["On the Measure of Intelligence","François Chollet","2019","paper","arXiv preprint","arxiv.org/abs/1911.01547",0,"","evals benchmarks training-data"],["Computing Receptive Fields of Convolutional Neural Networks","André Araujo and Wade Norris","2019","report","Distill","distill.pub/2019/computing-receptive-fields",0,"",""],["More variations on pseudo-alignment","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/iydwbZhATANhjoGP7/more-variations-on-pseudo-alignment",0,"","alignment-faking deception agents"],["Will transparency help catch deception? Perhaps not","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/J9D6Bi3eFDDhCaovi/will-transparency-help-catch-deception-perhaps-not",0,"","interpretability deception"],["But exactly how complex and fragile?","KatjaGrace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/xzFQp7bmkoKfnae9R/but-exactly-how-complex-and-fragile",0,"",""],["“embedded self-justification,” or something like that","nostalgebraist","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/kczouh3rvEoxJWFh5/embedded-self-justification-or-something-like-that",0,"","theory"],["AlphaStar: Impressive for RL progress, not for AGI progress","orthonormal","2019","blog","LessWrong","www.lesswrong.com/posts/SvhzEQkwFGNTy6CsN/alphastar-impressive-for-rl-progress-not-for-agi-progress",0,"","forecasting"],["Assessing the state of AI R&D in the US, China, and Europe – Part 1: Output indicators","stefan.torges","2019","blog","EA Forum","forum.effectivealtruism.org/posts/ng2h4EmCgaZK2GWF3/assessing-the-state-of-ai-r-and-d-in-the-us-china-and-europe",0,"","governance"],["Chris Olah’s views on AGI safety","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/X2i9dQQK3gETCyqh2/chris-olah-s-views-on-agi-safety",0,"","interpretability robustness"],["Generating Justifications for Norm-Related Agent Decisions","Daniel Kasenberg and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.00226",0,"","evals agents"],["Positive-Unlabeled Reward Learning","Danfei Xu and Misha Denil","2019","paper","arXiv preprint","arxiv.org/abs/1911.00459",0,"","agents"],["A Narration-based Reward Shaping Approach using Grounded Natural Language Commands","Nicholas Waytowich and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.00497",0,"","agents"],["Adversarial NLI: A New Benchmark for Natural Language Understanding","Yixin Nie","2019","paper","arXiv preprint","arxiv.org/abs/1910.14599",0,"","benchmarks"],["Conversation with Rohin Shah","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/conversation-with-rohin-shah/",0,"",""],["DeepLine: AutoML Tool for Pipelines Generation using Deep Reinforcement Learning and Hierarchical Actions Filtering","Yuval Heffetz and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.00061",0,"","evals agents"],["Rohin Shah on reasons for AI optimism","abergal","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TdwpN484eTbPSvZkm/rohin-shah-on-reasons-for-ai-optimism",0,"","interpretability"],["Rohin Shah on reasons for AI optimism","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/rohin-shah-on-reasons-for-ai-optimism/",0,"",""],["[AN #71]: Avoiding reward tampering through current-RF optimization","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jKBvMqs6t2RWAxWPc/an-71-avoiding-reward-tampering-through-current-rf",0,"","reward-hacking"],["Network Classifiers With Output Smoothing","Elsa Rizk and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.04870",0,"","agents"],["A Hamilton-Jacobi Reachability-Based Framework for Predicting and Analyzing Human Motion for Safe Planning","Somil Bansal and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.13369",0,"","agents"],["Doing Global Priorities or AI Policy research from remote location?","With Love from Israel","2019","blog","EA Forum","forum.effectivealtruism.org/posts/KpAa9uGoMY3b2htru/doing-global-priorities-or-ai-policy-research-from-remote",0,"","governance policy"],["October 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/10/25/october-2019-newsletter/",0,"",""],["Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","Tianhe Yu and 9 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.10897",0,"","evals benchmarks"],["[AN #70]: Agents that help humans who are still learning about their own preferences","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9mscdgJ7ao3vbbrjs/an-70-agents-that-help-humans-who-are-still-learning-about",0,"","agents"],["Deliberation as a method to find the \"actual preferences\" of humans","riceissa","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ebdf8GZxt3L9grwwN/deliberation-as-a-method-to-find-the-actual-preferences-of",0,"",""],["How can AI Automate End-to-End Data Science?","Charu Aggarwal and 11 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.14436",0,"",""],["Human-AI Collaboration","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/dBMC63hjkc5wPqTC7/human-ai-collaboration",0,"",""],["All I know is Goodhart","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/uL74oQv5PsnotGzt7/all-i-know-is-goodhart",0,"","goodharts-law"],["An Alternative Surrogate Loss for PGD-based Adversarial Testing","Sven Gowal and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.09338",0,"","benchmarks"],["Collaborating with Humans Requires Understanding Them","Rohin Shah and Micah Carroll","2019","report","bair.berkeley.edu","bair.berkeley.edu/blog/2019/10/21/coordination/",0,"",""],["Multi-agent Hierarchical Reinforcement Learning with Dynamic Termination","Dongge Han and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.09508",0,"","evals agents policy"],["The Psychology of Existential Risk: Moral Judgments about Human Extinction","Stefan Schubert and 2 others","2019","report","nature.com","www.nature.com/articles/s41598-019-50145-9",0,"",""],["The problem/solution matrix: Calculating the probability of AI safety \"on the back of an envelope\"","John_Maxwell","2019","blog","LessWrong","www.lesswrong.com/posts/znt3p9AGQDbYGf9Sy/the-problem-solution-matrix-calculating-the-probability-of",0,"",""],["[AN #69] Stuart Russell's new book on why we need to replace the standard model of AI","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/nd692YfFGfZDh9Mwz/an-69-stuart-russell-s-new-book-on-why-we-need-to-replace",0,"",""],["Defining Myopia","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qpZTWb2wvgSt5WQ4H/defining-myopia",0,"","theory"],["Summary of Stuart Russell's new book, \"Human Compatible\"","Rohin Shah","2019","blog","EA Forum","forum.effectivealtruism.org/posts/tsHfFdAGehzoH6BZR/summary-of-stuart-russell-s-new-book-human-compatible",0,"",""],["Jeffrey Ding: Re-deciphering China’s AI dream","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/JgW894h7yWgnB6bT4/jeffrey-ding-re-deciphering-china-s-ai-dream",0,"","governance policy"],["Jesse Clifton: Open-source learning — a bargaining approach","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/YnRYMLxaw43daKmeT/jesse-clifton-open-source-learning-a-bargaining-approach",0,"",""],["Ross Gruetzemacher: Defining and unpacking transformative AI","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/LywpuDpNEhTw8iqR3/ross-gruetzemacher-defining-and-unpacking-transformative-ai",0,"","governance"],["Technical AGI safety research outside AI","Richard_Ngo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4xbsi4wbourPkb47x/technical-agi-safety-research-outside-ai",0,"","forecasting"],["Technical AGI safety research outside AI","richard_ngo","2019","blog","EA Forum","forum.effectivealtruism.org/posts/2e9NDGiXt8PjjbTMC/technical-agi-safety-research-outside-ai",0,"","forecasting"],["Random Thoughts on Predict-O-Matic","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/25288usP5B5ytnzA4/random-thoughts-on-predict-o-matic",0,"",""],["The Dualist Predict-O-Matic ($100 prize)","John_Maxwell","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/RmPKdMqSr2xRwrqyE/the-dualist-predict-o-matic-usd100-prize",0,"",""],["Full toy model for preference learning","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hcrFxeYYfbFrkKQEJ/full-toy-model-for-preference-learning",0,"",""],["Gradient hacking","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/uXH4r6MmKPedk8rMA/gradient-hacking",0,"","interpretability agents"],["Impact measurement and value-neutrality verification","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jGB7Pd5q8ivBor8Ee/impact-measurement-and-value-neutrality-verification-1",0,"",""],["Restoring ancient text using deep learning: a case study on Greek epigraphy","Yannis Assael and 2 others","2019","blog","deepmind.com","www.deepmind.com/blog/restoring-ancient-text-using-deep-learning-a-case-study-on-greek-epigraphy",0,"",""],["Solving Logic Grid Puzzles with an Algorithm that Imitates Human Behavior","Guillaume Escamocher and Barry O'Sullivan","2019","paper","arXiv preprint","arxiv.org/abs/1910.06636",0,"",""],["The Parable of Predict-O-Matic","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SwcyMEgLyd4C3Dern/the-parable-of-predict-o-matic",0,"",""],["[AN #68]: The attainable utility theory of impact","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HeuJZfexbTBRijTs2/an-68-the-attainable-utility-theory-of-impact",0,"",""],["Restoration of marker occluded hematoxylin and eosin stained whole slide histology images using generative adversarial networks","Bairavi Venkatesh and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.06428",0,"","deception"],["Using AI/ML to gain situational understanding from passive network observations","D. Verma and S. Calo","2019","paper","arXiv preprint","arxiv.org/abs/1910.06266",0,"","situational-awareness policy"],["AI alignment landscape","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zQgqAc9nQETjFjawJ/ai-alignment-landscape",0,"",""],["On the Utility of Learning about Humans for Human-AI Coordination","Micah Carroll and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.05789",0,"","evals agents robustness"],["Stabilizing Transformers for Reinforcement Learning","Emilio Parisotto and 12 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.06764",0,"","evals benchmarks agents"],["Asking Easy Questions: A User-Friendly Approach to Active Reward Learning","Erdem Bıyık and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.04365",0,"",""],["Imitation Learning from Observations by Minimizing Inverse Dynamics Disagreement","Chao Yang and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.04417",0,"","benchmarks"],["The Quest for Interpretable and Responsible Artificial Intelligence","Vaishak Belle","2019","paper","arXiv preprint","arxiv.org/abs/1910.04527",0,"","interpretability"],["Thoughts on \"Human-Compatible\"","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/FuGDYNvA6qh4qyFah/thoughts-on-human-compatible",0,"",""],["Improving Generalization in Meta Reinforcement Learning using Learned Objectives","Louis Kirsch and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.04098",0,"","agents policy"],["Integrating Behavior Cloning and Reinforcement Learning for Improved Performance in Dense and Sparse Reward Environments","Vinicius G. Goecks and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.04281",0,"","policy"],["Minimization of prediction error as a foundation for human values in AI alignment","Gordon Seidoh Worley","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Cu7yv4eM6dCeA67Af/minimization-of-prediction-error-as-a-foundation-for-human",0,"",""],["Can We Distinguish Machine Learning from Human Learning?","Vicki Bier and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.03466",0,"",""],["Characterizing Real-World Agents as a Research Meta-Strategy","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9pZtvjegYKBALFnLk/characterizing-real-world-agents-as-a-research-meta-strategy",0,"","agents"],["Detecting AI Trojans Using Meta Neural Analysis","Xiaojun Xu  Qi Wang  Huichen Li  Nikita Borisov  Carl A. Gunter  Bo Li","2019","paper","arXiv preprint","arxiv.org/abs/1910.03137",0,"","evals robustness training-data"],["Human Compatible: Artificial Intelligence and the Problem of Control","Stuart Russell","2019","report","goodreads.com","www.goodreads.com/book/show/44767248-human-compatible",0,"",""],["Misconceptions about continuous takeoff","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/CjW4axQDqLd2oDCGG/misconceptions-about-continuous-takeoff",0,"","agents forecasting"],["What's the dream for giving natural language commands to AI?","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Bxxh9GbJ6WuW5Hmkj/what-s-the-dream-for-giving-natural-language-commands-to-ai",0,"","goodharts-law agents"],["[AN #67]: Creating environments in which to study inner alignment failures","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SLHQF25vRXoNEooSG/an-67-creating-environments-in-which-to-study-inner",0,"",""],["AI Alignment Writing Day Roundup #2","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/DG7asvufKgaqEknKd/ai-alignment-writing-day-roundup-2",0,"",""],["Occam's Razor May Be Sufficient to Infer the Preferences of Irrational Agents: A reply to Armstrong & Mindermann","Daniel Kokotajlo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pHWTNMESuAEjZg2Qn/occam-s-razor-may-be-sufficient-to-infer-the-preferences-of",0,"","agents"],["The Gears of Impact","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/coQCEe962sjbcCqB9/the-gears-of-impact",0,"",""],["Towards Deployment of Robust AI Agents for Human-Machine Partnerships","Ahana Ghosh and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.02330",0,"","agents policy"],["AI Alignment Open Thread October 2019","habryka","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pFAavCTW56iTsYkvR/ai-alignment-open-thread-october-2019",0,"",""],["Debate on Instrumental Convergence between LeCun, Russell, Bengio, Zador, and More","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WxW6Gc6f2z3mzmqKs/debate-on-instrumental-convergence-between-lecun-russell",0,"","instrumental-convergence"],["The AI is the model","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HS2E8woaF5h5QSptP/the-ai-is-the-model",0,"",""],["Universality and model-based RL","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/universality-and-model-based-rl-b08701394ddd",0,"","scalable-oversight evals agents robustness"],["Can we make peace with moral indeterminacy?","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qpJbFta7RwpHcFarc/can-we-make-peace-with-moral-indeterminacy",0,"",""],["Formal Language Constraints for Markov Decision Processes","Eleanor Quint and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.01074",0,"","evals deception agents"],["Human instincts, symbol grounding, and the blank-slate neocortex","Steven Byrnes","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NkSpukDkm9pjRdMdB/human-instincts-symbol-grounding-and-the-blank-slate",0,"",""],["Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","Denis Yarats and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.01741",0,"","deception agents policy robustness"],["AI Alignment Research Overview","Jacob Steinhardt","2019","report","docs.google.com","docs.google.com/document/d/1FbTuRvC4TFWzGYerTKpBU7FJlyvjeOvVYF2uYNFSlOc/edit#heading=h.n1wk9bxo847o",0,"",""],["Doing more with less: meta-reasoning and meta-learning in humans and machines","Thomas L Griffiths and 5 others","2019","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S2352154618302122",0,"",""],["It's not too soon to be wary of AI: We need to act now to protect humanity from future superintelligent machines","Stuart Russell","2019","report","doi.org","doi.org/10.1109/MSPEC.2019.8847590",0,"",""],["September 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/09/30/september-2019-newsletter/",0,"",""],["World State is the Wrong Abstraction for Impact","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pr3bLc2LtjARfK7nx/world-state-is-the-wrong-abstraction-for-impact",0,"",""],["[AN #66]: Decomposing robustness into capability robustness and alignment robustness","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Epm6CkXrdRyAihMRe/an-66-decomposing-robustness-into-capability-robustness-and",0,"","robustness"],["List of resolved confusions about IDA","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/FdfzFcRvqLf4k5eoQ/list-of-resolved-confusions-about-ida",0,"",""],["The Paths Perspective on Value Learning","Sam Greydanus and Chris Olah","2019","report","Distill","distill.pub/2019/paths-perspective-on-value-learning",0,"",""],["Christiano decision theory excerpt","Rob Bensinger","2019","blog","LessWrong","www.lesswrong.com/posts/n6wajkE3Tpfn6sd5j/christiano-decision-theory-excerpt",0,"","theory"],["Gradient Descent: The Ultimate Optimizer","Kartik Chandra and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.13371",0,"",""],["Learning from Observations Using a Single Video Demonstration and Human Feedback","Sunil Gandhi and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.13392",0,"","rlhf agents policy"],["UK policy and politics careers","weeatquince","2019","blog","EA Forum","forum.effectivealtruism.org/posts/5nPo6nPYZz4F2h5sa/uk-policy-and-politics-careers",0,"","governance policy"],["[Talk] Paul Christiano on his alignment taxonomy","jp","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/S4MsqB7TRsoWLBmfZ/talk-paul-christiano-on-his-alignment-taxonomy",0,"",""],["A Constructive Prediction of the Generalization Error Across Scales","Jonathan S. Rosenfeld and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.12673",0,"","robustness"],["Attainable Utility Theory: Why Things Matter","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/C74F7QTEAYSTGAytJ/attainable-utility-theory-why-things-matter",0,"",""],["Automated curricula through setter-solver interactions","Sebastien Racaniere and 5 others","2019","paper","International Conference on Learning Representations, 2020","arxiv.org/abs/1909.12892",0,"","deception agents"],["Partial Agency","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4hdHto3uHejhY2F3Q/partial-agency",0,"",""],["A simple environment for showing mesa misalignment","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/AFdRGfYDWQqmkdhFq/a-simple-environment-for-showing-mesa-misalignment",0,"","robustness"],["Scaling data-driven robotics with reward sketching and batch reinforcement learning","Serkan Cabi and 15 others","2019","paper","Robotics: Science and Systems Conference 2020","arxiv.org/abs/1909.12200",0,"","agents"],["Toward Evaluating Robustness of Deep Reinforcement Learning with Continuous Control","Tsui-Wei Weng and 6 others","2019","report","openreview.net","openreview.net/forum?id=SylL0krYPS",0,"","evals robustness"],["Deducing Impact","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qs88fvwnjBevMrbkK/deducing-impact",0,"",""],["[AN #65]: Learning useful skills by watching humans “play”","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/GPADepj6yP8zqSbJh/an-65-learning-useful-skills-by-watching-humans-play",0,"",""],["Towards an empirical investigation of inner alignment","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/2GycxikGnepJbxfHT/towards-an-empirical-investigation-of-inner-alignment",0,"","agents robustness"],["Value Impact","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TxcYSRQ9giC6zmKov/value-impact",0,"",""],["Scaled Autonomy: Enabling Human Operators to Control Robot Fleets","Gokul Swamy and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.02910",0,"","evals"],["Leveraging Human Guidance for Deep Reinforcement Learning Tasks","Ruohan Zhang and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.09906",0,"","agents"],["What are the differences between all the iterative/recursive approaches to AI alignment?","riceissa","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cYduioQNeHALQAMre/what-are-the-differences-between-all-the-iterative-recursive",0,"","scalable-oversight"],["Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","Lantao Yu and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.09314",0,"","agents"],["Preparing for the unthinkable","Seth D. Baum","2019","report","sciencemag.org","www.sciencemag.org/lookup/doi/10.1126/science.aay4219",0,"",""],["Reframing Impact","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/xCxeBSHqMEaP3jDvY/reframing-impact",0,"",""],["How does the offense-defense balance scale?","Ben Garfinkel and Allan Dafoe","2019","report","doi.org","doi.org/10.1080/01402390.2019.1631810",0,"",""],["Fine-Tuning Language Models from Human Preferences","Daniel M. Ziegler and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.08593",0,"","evals robustness"],["The unexpected difficulty of comparing AlphaStar to humans","richardkorzekwa","2019","blog","aiimpacts.org","aiimpacts.org/the-unexpected-difficulty-of-comparing-alphastar-to-humans/",0,"",""],["The unexpected difficulty of comparing AlphaStar to humans","Richard Korzekwa","2019","blog","LessWrong","www.lesswrong.com/posts/FpcgSoJDNNEZ4BQfj/the-unexpected-difficulty-of-comparing-alphastar-to-humans",0,"","forecasting"],["Emergent Tool Use From Multi-Agent Autocurricula","Bowen Baker and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.07528",0,"","evals agents tool-use"],["[AN #64]: Using Deep RL and Reward Uncertainty to Incentivize Preference Learning","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/JxWggFXcKrPKy7p8t/an-64-using-deep-rl-and-reward-uncertainty-to-incentivize",0,"",""],["Neural Cleanse: Identifying and Mitigating Backdoor Attacks in Neural Networks","Bolun Wang and 6 others","2019","report","people.cs.uchicago.edu","people.cs.uchicago.edu/~ravenben/publications/pdf/backdoor-sp19.pdf",0,"",""],["Realism and Rationality","bmgarfinkel","2019","blog","LessWrong","www.lesswrong.com/posts/GqTeChFnXdJzDzbMd/realism-and-rationality-2",0,"","theory"],["The strategy-stealing assumption","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/nRAMpjnb6Z4Qv3imF/the-strategy-stealing-assumption",0,"","ai-control"],["The strategy-stealing assumption","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/the-strategy-stealing-assumption-a26b8b1ed334",0,"",""],["VILD: Variational Imitation Learning with Diverse-quality Demonstrations","Voot Tangkaratt and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.06769",0,"","benchmarks policy robustness"],["A Critique of Functional Decision Theory","wdmacaskill","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ySLYSsNeFL5CoAQzN/a-critique-of-functional-decision-theory",0,"","agents theory"],["Do Sufficiently Advanced Agents Use Logic?","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/3qXE6fK47JhSfkpnB/do-sufficiently-advanced-agents-use-logic",0,"","agents"],["What You See Isn't Always What You Want","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/AeHtdxHheMjHredaq/what-you-see-isn-t-always-what-you-want",0,"","scalable-oversight deception agents robustness"],["Better AI through Logical Scaffolding","Nikos Arechiga and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.06965",0,"","agents"],["Finding Generalizable Evidence by Learning to Convince Q&A Models","Ethan Perez and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.05863",0,"","debate agents"],["Conversation with Paul Christiano","abergal","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qnYZmtpNPZyqHpot9/conversation-with-paul-christiano",0,"",""],["Conversation with Paul Christiano","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/conversation-with-paul-christiano/",0,"","evals robustness"],["Paul Christiano on the safety of future AI systems","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/paul-christiano-on/",0,"",""],["Soft takeoff can still lead to decisive strategic advantage","Daniel Kokotajlo","2019","blog","aiimpacts.org","aiimpacts.org/soft-takeoff-can-still-lead-to-decisive-strategic-advantage/",0,"","forecasting"],["[AN #63] How architecture search, meta learning, and environment design could lead to general intelligence","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/E9G5JYbXZ3QK9aXTm/an-63-how-architecture-search-meta-learning-and-environment",0,"",""],["Counterfactual Oracles = online supervised learning with random selection of training episodes","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yAiqLmLFxvyANSfs2/counterfactual-oracles-online-supervised-learning-with",0,"",""],["Is my result wrong? Maths vs intuition vs evolution in learning human preferences","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/2KLz6RQWkCj4Rozrk/is-my-result-wrong-maths-vs-intuition-vs-evolution-in",0,"",""],["Meta-Learning with Implicit Gradients","Aravind Rajeswaran and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.04630",0,"","agents"],["Relaxed adversarial training for inner alignment","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9Dy5YRaoCxH9zuJqa/relaxed-adversarial-training-for-inner-alignment",0,"","scalable-oversight interpretability"],["AI Safety \"Success Stories\"","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/bnY3L48TtDrKTzGRb/ai-safety-success-stories",0,"",""],["Are minimal circuits deceptive?","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/fM5ZWGDbnjb7ThNKJ/are-minimal-circuits-deceptive",0,"","mechanistic-interpretability deception policy robustness"],["Concrete experiments in inner alignment","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/uSdPa9nrSgmXCtdKN/concrete-experiments-in-inner-alignment",0,"","agents"],["How much EA analysis of AI safety as a cause area exists?","richard_ngo","2019","blog","EA Forum","forum.effectivealtruism.org/posts/u3ePLsbtpkmFdD7Nb/how-much-ea-analysis-of-ai-safety-as-a-cause-area-exists-1",0,"",""],["How to Throw Away Information","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/DEDcFw6zWfW9nb2YM/how-to-throw-away-information",0,"",""],["Implications of Quantum Computing for Artificial Intelligence alignment research (ABRIDGED)","Jaime Sevilla","2019","blog","EA Forum","forum.effectivealtruism.org/posts/meTqCDCNzYgYmkF76/implications-of-quantum-computing-for-artificial",0,"",""],["Logical Counterfactuals and Proposition graphs, Part 3","Donald Hobson","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/fhbb8MGEs3t5dTCLD/logical-counterfactuals-and-proposition-graphs-part-3",0,"",""],["Making Efficient Use of Demonstrations to Solve Hard Exploration Problems","Caglar Gülçehre and 13 others","2019","blog","deepmind.com","www.deepmind.com/blog/making-efficient-use-of-demonstrations-to-solve-hard-exploration-problems",0,"",""],["Utility ≠ Reward","vlad_m","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/bG4PR9uSsZqHg2gYY/utility-reward",0,"",""],["Utility ≠ Reward","Vlad Mikulik","2019","blog","LessWrong","www.lesswrong.com/posts/bG4PR9uSsZqHg2gYY/utility-reward",0,"",""],["Achieving Verified Robustness to Symbol Substitutions via Interval Bound Propagation","Po-Sen Huang and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.01492",0,"","robustness"],["AI Forecasting Question Database (Forecasting infrastructure, part 3)","jacobjacob and goldhaber","2019","blog","EA Forum","forum.effectivealtruism.org/posts/eYkr8x9QTzgFs7mu7/ai-forecasting-question-database-forecasting-infrastructure",0,"","forecasting"],["Authoritarian Audiences, Rhetoric, and Propaganda in International Crises: Evidence from China","Jessica Chen Weiss and Allan Dafoe","2019","report","academic.oup.com","academic.oup.com/isq/article/63/4/963/5559531",0,"",""],["Counterfactuals are an Answer, Not a Question","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ao7KLoBEvMdHFjrNZ/counterfactuals-are-an-answer-not-a-question",0,"",""],["LCA: Loss Change Allocation for Neural Network Training","Janice Lan and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.01440",0,"",""],["Making Efficient Use of Demonstrations to Solve Hard Exploration Problems","Tom Le Paine and 13 others","2019","paper","arXiv preprint","arxiv.org/abs/1909.01387",0,"","agents"],["Probability as Minimal Map","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Lz2nCYnBeaZyS68Xb/probability-as-minimal-map",0,"",""],["Shortcomings of the Bow Tie and Other Safety Tools Based on Linear Causality","Prof. Nancy G. Leveson","2019","report","sunnyday.mit.edu","sunnyday.mit.edu/Bow-tie-final.pdf",0,"",""],["Logical Counterfactuals and Proposition graphs, Part 2","Donald Hobson","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/due5BtsbpTzSZbeKT/logical-counterfactuals-and-proposition-graphs-part-2",0,"",""],["2-D Robustness","vlad_m","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/2mhFMgtAjFJesaSYR/2-d-robustness",0,"","agents robustness"],["2-D Robustness","Vlad Mikulik","2019","blog","LessWrong","www.lesswrong.com/posts/2mhFMgtAjFJesaSYR/2-d-robustness",0,"","agents robustness"],["AI Alignment Writing Day Roundup #1","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZYGjDpGQaHvg8HLfw/ai-alignment-writing-day-roundup-1",0,"",""],["From the Internet of Information to the Internet of Intelligence","F. Richard Yu","2019","paper","arXiv preprint","arxiv.org/abs/1909.08068",0,"","training-data"],["judiciary.senate.gov","Caspar Oesterheld and Vincent Conitzer","2019","report","ceur-ws.org","ceur-ws.org/Vol-2640/paper_21.pdf",0,"",""],["AI Forecasting Resolution Council (Forecasting infrastructure, part 2)","jacobjacob and goldhaber","2019","blog","EA Forum","forum.effectivealtruism.org/posts/QZRxWqJHYcZnzSJTf/ai-forecasting-resolution-council-forecasting-infrastructure",0,"","forecasting"],["[Link] Book Review: Reframing Superintelligence (SSC)","ioannes","2019","blog","LessWrong","www.lesswrong.com/posts/wkF5rHDFKEWyJJLj2/link-book-review-reframing-superintelligence-ssc",0,"",""],["How can we see the impact of AI strategy research _ Jade Leung _ EA Global - San Francisco 2019-by Centre for Effective Altruism-video_id 8M3nIu7GIsA-date 20190829","Jade Leung","2019","report","drive.google.com","drive.google.com/file/d/1wCQyIFbCd08d2OII9KyGJGsNq7AfVvdg/view?usp=share_link",0,"",""],["Sino-Western cooperation in AI safety _ Brian Tse _ EA Global - San Francisco 2019-by Centre for Effective Altruism-video_id 3qYmLRqemg4-date 20190829","Brian Tse","2019","report","drive.google.com","drive.google.com/file/d/10BV6iaQ59OQ0y3cKYTqmg31inPnabNRg/view?usp=share_link",0,"",""],["The Windfall Clause - Sharing the benefits of advanced AI _ Cullen OΓÇÖKeefe-by Centre for Effective Altruism-video_id vFDL-NxY610-date 20190829","Cullen O'Keefe","2019","report","drive.google.com","drive.google.com/file/d/1tOKaR_9chGQFePFXUBPalyAgG78rQ8z9/view?usp=share_link",0,"",""],["Training machine learning (ML) systems to answer open-ended questions _ Andreas Stuhlmuller-by Centre for Effective Altruism-video_id 7WaiYZLS94M-date 20190829","Andreas Stuhlmüller","2019","report","drive.google.com","drive.google.com/file/d/1i_SHxE0Cn-UelJ9Z9fGVcFoYtbdeJCdu/view?usp=share_link",0,"",""],["AI & Policy 1/3: On knowing the effect of today’s policies on Transformative AI risks, and the case for institutional improvements.","weeatquince","2019","blog","EA Forum","forum.effectivealtruism.org/posts/jMyjwRMMkYCnFmMHH/ai-and-policy-1-3-on-knowing-the-effect-of-today-s-policies",0,"","governance policy"],["Cartographic Processes","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/t5DFpygMqpnFsmJ3b/cartographic-processes",0,"",""],["End Times: A Brief Guide to the End of the World","Bryan Walsh","2019","report","goodreads.com","www.goodreads.com/book/show/42283306-end-times",0,"",""],["Six AI Risk/Strategy Ideas","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/dt4z82hpvvPFTDTfZ/six-ai-risk-strategy-ideas",0,"",""],["Building The Castle vs Finding The Monolith • carado.moe","Tamsin Leake","2019","blog","carado.moe","carado.moe/castle-monolith.html",0,"",""],["Embedded Agency via Abstraction","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hLFD6qSN9MmQxKjG5/embedded-agency-via-abstraction",0,"","deception agents theory"],["Problems with AI debate","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/fNTCveSa4HvqvZR2F/problems-with-ai-debate",0,"",""],["Reversible changes: consider a bucket of water","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zrunBA8B5bmm2XZ59/reversible-changes-consider-a-bucket-of-water",0,"",""],["Gratification: a useful concept, maybe new","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/GnPSQAi3QzHjK8ZQR/gratification-a-useful-concept-maybe-new",0,"",""],["Under a week left to win $1,000! By questioning Oracle AIs.","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/bnDGt4Y9Hfx62Bgmk/under-a-week-left-to-win-usd1-000-by-questioning-oracle-ais",0,"",""],["Ernie Davis on the landscape of AI risks","Rob Long","2019","blog","aiimpacts.org","aiimpacts.org/ernie-davis-on-the-landscape-of-ai-risks/",0,"",""],["Release Strategies and the Social Impacts of Language Models","Irene Solaiman and 14 others","2019","paper","arXiv preprint","arxiv.org/abs/1908.09203",0,"",""],["Algorithmic Similarity","LukasM","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BS7Syu2buhLYRjkwY/algorithmic-similarity",0,"",""],["Conversation with Ernie Davis","Rob Long","2019","blog","aiimpacts.org","aiimpacts.org/conversation-with-ernie-davis/",0,"",""],["Creating Environments to Design and Test Embedded Agents","lukehmiles","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qdqYrcGZTh9Lp49Nj/creating-environments-to-design-and-test-embedded-agents",0,"","agents"],["Does Agent-like Behavior Imply Agent-like Architecture?","Scott Garrabrant","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/osxNg6yBCJ4ur9hpi/does-agent-like-behavior-imply-agent-like-architecture",0,"","agents"],["Existential risks: a philosophical analysis","Phil Torres","2019","report","doi.org","doi.org/10.1080/0020174X.2019.1658626",0,"",""],["Formalising decision theory is hard","Lukas Finnveden","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/S3W4Xrmp6AL7nxRHd/formalising-decision-theory-is-hard",0,"","theory"],["Is there a simple parameter that controls human working memory capacity, which has been set tragically low?","Liron","2019","blog","LessWrong","www.lesswrong.com/posts/NptgfCiJvXyoRgdcz/is-there-a-simple-parameter-that-controls-human-working",0,"",""],["Optimization Provenance","Adele Lopez","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Zj2PgP5A8vY2G3gYw/optimization-provenance",0,"",""],["Soft takeoff can still lead to decisive strategic advantage","Daniel Kokotajlo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/PKy8NuNPknenkDY74/soft-takeoff-can-still-lead-to-decisive-strategic-advantage",0,"","governance forecasting"],["Tabooing 'Agent' for Prosaic Alignment","Hjalmar_Wijk","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zCcmJzbenAXu6qugS/tabooing-agent-for-prosaic-alignment",0,"","agents"],["The Commitment Races problem","Daniel Kokotajlo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/brXr7PJ2W4Na2EW2q/the-commitment-races-problem",0,"","agents"],["Thoughts from a Two Boxer","jaek","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9QztnkMiKJ7jYZhL8/thoughts-from-a-two-boxer",0,"","theory"],["Towards an Intentional Research Agenda","romeostevensit","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/CHSRhSKcrSmQWnD6A/towards-an-intentional-research-agenda",0,"",""],["Troll Bridge","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hpAbfXtqYC2BrpeiC/troll-bridge-5",0,"","agents theory"],["Understanding understanding","mthq","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WGE6nqbWzzDxRmMs7/understanding-understanding",0,"","interpretability"],["Vaniver's View on Factored Cognition","Vaniver","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/J7Rnt8aJPH7MALkmq/vaniver-s-view-on-factored-cognition",0,"",""],["When do utility functions constrain?","Hoagy","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yGuo5R9fgrrFLYWuv/when-do-utility-functions-constrain-1",0,"",""],["[AN #62] Are adversarial examples caused by real but imperceptible features?","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NTwA3J99RPkgmp6jh/an-62-are-adversarial-examples-caused-by-real-but",0,"","robustness"],["Announcement: Writing Day Today (Thursday)","Ben Pace","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/aJGXWHnYTsWwAiwf6/announcement-writing-day-today-thursday",0,"",""],["Computational Model: Causal Diagrams with Symmetry","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/mZy6AMgCw9CPjNCoK/computational-model-causal-diagrams-with-symmetry",0,"",""],["Implications of Quantum Computing for Artificial Intelligence Alignment Research","Jsevillamol and PabloAMC","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZkgqsyWgyDx4ZssqJ/implications-of-quantum-computing-for-artificial",0,"","forecasting"],["Logical Counterfactuals and Proposition graphs, Part 1","Donald Hobson","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Pxvq2RMAKCuY6SHm9/logical-counterfactuals-and-proposition-graphs-part-1",0,"",""],["Markets are Universal for Logical Induction","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WmNeCipNwg9CmGy3T/markets-are-universal-for-logical-induction",0,"","theory"],["Towards a mechanistic understanding of corrigibility","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BKM8uQS6QdJPZLqCr/towards-a-mechanistic-understanding-of-corrigibility",0,"","scalable-oversight"],["Call for contributors to the Alignment Newsletter","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cejX4S3Dex3C2dt79/call-for-contributors-to-the-alignment-newsletter",0,"",""],["Testing Robustness Against Unforeseen Adversaries","Daniel Kang and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1908.08016",0,"","evals benchmarks robustness"],["Two senses of “optimizer”","Joar Skalse","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/rvxcSc6wdcCfaX6GZ/two-senses-of-optimizer",0,"",""],["Universal Adversarial Triggers for Attacking and Analyzing NLP WARNING: This paper contains model outputs which are offensive in nature.","","2019","paper","arXiv preprint","arxiv.org/abs/1908.07125",0,"","evals robustness"],["Classifying specification problems as variants of Goodhart's Law","Vika","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXPT4nr4as7JvxLQa/classifying-specification-problems-as-variants-of-goodhart-s",0,"","reward-hacking goodharts-law assurance robustness"],["Classifying specification problems as variants of Goodhart’s Law","Victoria Krakovna","2019","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2019/08/19/classifying-specification-problems-as-variants-of-goodharts-law/",0,"","goodharts-law"],["Goodhart's Curse and Limitations on AI Alignment","Gordon Seidoh Worley","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NqQxTn5MKEYhSnbuB/goodhart-s-curse-and-limitations-on-ai-alignment",0,"","goodharts-law"],["Implications of Quantum Computing for Artificial Intelligence alignment research","Jaime Sevilla and Pablo Moreno","2019","paper","arXiv preprint","arxiv.org/abs/1908.07613",0,"",""],["Problems in AI Alignment that philosophers could potentially contribute to","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/rASeoR7iZ9Fokzh7L/problems-in-ai-alignment-that-philosophers-could-potentially",0,"","forecasting theory"],["Clarifying some key hypotheses in AI alignment","Ben Cottier and Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/mJ5oNYnkYrd4sD5uE/clarifying-some-key-hypotheses-in-ai-alignment",0,"",""],["Hard Choices in Artificial Intelligence: Addressing Normative Uncertainty through Sociotechnical Commitments.","Roel Dobbe and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1911.09005",0,"","evals"],["Legible Normativity for AI Alignment: The Value of Silly Rules.","Dylan Hadfield-Menell and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1811.01267",0,"","agents"],["Literal or Pedagogic Human? Analyzing Human Model Misspecification in Objective Learning.","Smitha Milli and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.03877",0,"","robustness"],["On the Utility of Model Learning in HRI.","Rohan Choudhury and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.01291",0,"","policy"],["Reward-rational (implicit) choice: A unifying formalism for reward learning.","Hong Jun Jeon and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/2002.04833",0,"","deception agents robustness"],["The Assistive Multi-Armed Bandit.","Lawrence Chan and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.08654",0,"","agents"],["A Parametric, Resource-Bounded Generalization of Löb’s Theorem, and a Robust Cooperation Criterion for Open-Source Game Theory.","Andrew Critch","2019","report","cambridge.org","www.cambridge.org/core/journals/journal-of-symbolic-logic/article/parametric-resourcebounded-generalization-of-lobs-theorem-and-a-robust-cooperation-criterion-for-opensource-game-theory/16063EA7BFFEE89438631B141E556E79#",0,"",""],["A Risk-Sensitive Finite-Time Reachability Approach for Safety of Stochastic Dynamic Systems.","Margaret P and 13 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.11277",0,"","policy"],["A unified framework for planning in adversarial and cooperative environments.","Anagha Kulkarni and 2 others","2019","report","aaai.org","www.aaai.org/ojs/index.php/AAAI/article/view/4093",0,"",""],["Abstracting causal models.","Sander Beckers and 2 others","2019","report","m.aaai.org","m.aaai.org/ojs/index.php/AAAI/article/view/4117",0,"",""],["Adversarial Training with Voronoi Constraints.","Marc Khoury and Dylan Hadfield-Menell","2019","paper","arXiv preprint","arxiv.org/abs/1905.01019",0,"","robustness"],["An Agent-Based Model of Financial Benchmark Manipulation.","Megan Shearer and 3 others","2019","report","par.nsf.gov","par.nsf.gov/biblio/10105527",0,"","benchmarks agents"],["Approximate Causal Abstraction.","Sander Beckers and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.11583",0,"","deception"],["Bayesian Robustness: A Nonasymptotic Viewpoint.","Kush Bhatia and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.11826",0,"","robustness"],["Benchmarking Neural Network Robustness to Common Corruptions and Perturbations.","Dan Hendrycks and Thomas Dietterich","2019","paper","arXiv preprint","arxiv.org/abs/1903.12261",0,"","evals benchmarks robustness"],["Blameworthiness in Multi-Agent Settings.","Meir Friedenberg and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.04102",0,"","agents robustness"],["Bridging Hamilton-Jacobi Safety Analysis and Reinforcement Learning.","Jaime F and 7 others","2019","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~jfisac/papers/Bridging_Safety_and_RL.pdf",0,"",""],["Capturing human categorization of natural images at scale by combining deep networks and cognitive models.","Ruairidh M and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.12690",0,"","deception"],["Causal Discovery in the Presence of Missing Data.","Ruibo Tu and 5 others","2019","report","proceedings.mlr.press","proceedings.mlr.press/v89/tu19a.html",0,"",""],["Deep Anomaly Detection with Outlier Exposure.","Dan Hendrycks and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1812.04606",0,"","robustness monitoring"],["Distributed Protocols for Leader Election: A Game-Theoretic Perspective.","Ittai Abraham and 3 others","2019","report","dl.acm.org","dl.acm.org/citation.cfm?id=3303712",0,"",""],["Epistemic Therapy for Bias in Automated Decision-Making.","Thomas Krendl Gilbert and Yonatan Mintz","2019","report","dl.acm.org","dl.acm.org/doi/abs/10.1145/3306618.3314294",0,"",""],["Graphical Models for Processing Missing Data.","Karthika Mohan and Judea Pearl","2019","paper","arXiv preprint","arxiv.org/abs/1801.03583",0,"","interpretability"],["Hierarchically Decoupled Imitation for Morphological Transfer.","Donald J and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/2003.01709",0,"","agents policy"],["How You Act Tells a Lot: Privacy-Leaking Attack on Deep Reinforcement Learning.","Xinlei Pan and 5 others","2019","report","dl.acm.org","dl.acm.org/citation.cfm?id=3331715",0,"",""],["Human Compatible: Artificial Intelligence and The Problem of Control.","Stuart Russell","2019","report","penguinrandomhouse.com","www.penguinrandomhouse.com/books/566677/human-compatible-by-stuart-russell/",0,"",""],["Implementing Mediators with Asynchronous Cheap Talk.","Ittai Abraham and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1806.01214",0,"","agents"],["Incentivizing Collaboration in a Competition.","Arunesh Sinha and 2 others","2019","report","dl.acm.org","dl.acm.org/citation.cfm?id=3331740",0,"",""],["Learning a Prior over Intent via Meta-Inverse Reinforcement Learning.","Kelvin Xu and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1805.12573",0,"",""],["Learning-Based Trading Strategies in the Face of Market Manipulation.","Xintong Wang and 3 others","2019","report","par.nsf.gov","par.nsf.gov/biblio/10105525",0,"",""],["Natural Adversarial Examples.","Dan Hendrycks and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.07174",0,"","robustness training-data"],["On the Existence of Nash Equilibrium in Games with Resource-Bounded Players.","Joseph Y and 3 others","2019","report","link.springer.com","link.springer.com/chapter/10.1007/978-3-030-30473-7_10",0,"",""],["Partial Awareness.","Joseph Y and 2 others","2019","report","pdfs.semanticscholar.org","pdfs.semanticscholar.org/d1d8/0d86fb46b46129bc97d817ba1f0e34bada63.pdf",0,"",""],["Quantifying Hypothesis Space Misspecification in Learning from Human-Robot Demonstrations and Physical Corrections.","Andreea Bobu and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/2002.00941",0,"","agents"],["Resource-rational analysis: understanding human cognition as the optimal use of limited computational resources.","Falk Lieder and 2 others","2019","report","cambridge.org","www.cambridge.org/core/journals/behavioral-and-brain-sciences/article/resourcerational-analysis-understanding-human-cognition-as-the-optimal-use-of-limited-computational-resources/586866D9AD1D1EA7A1EECE217D392F4A",0,"",""],["Security in Asynchronous Interactive Systems.","Ivan Geffner and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.02069",0,"","agents"],["Sequential equilibrium in computational games.","Joseph Y and 2 others","2019","report","dl.acm.org","dl.acm.org/citation.cfm?id=3340232",0,"",""],["Strategic Classification is Causal Modeling in Disguise.","John Miller and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1910.10362",0,"",""],["The Computational Structure of Unintentional Meaning.","Mark K and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.01983",0,"",""],["Towards a Just Theory of Measurement: A Principled Social Measurement Assurance Program for Machine Learning.","McKane Andrus and Thomas Krendl Gilbert","2019","report","dl.acm.org","dl.acm.org/doi/abs/10.1145/3306618.3314275",0,"","assurance"],["Using Machine Learning to Guide Cognitive Modeling: A Case Study in Moral Reasoning.","Mayank Agrawal and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.06744",0,"","interpretability"],["Using Self-Supervised Learning Can Improve Model Robustness and Uncertainty.","Dan Hendrycks and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.12340",0,"","evals robustness"],["Distance Functions are Hard","Grue_Slinky","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/YuJNoCEgeWJfBtdtQ/distance-functions-are-hard-1",0,"","theory"],["Evidence against current methods leading to human level artificial intelligence","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/evidence-against-current-methods-leading-to-human-level-artificial-intelligence/",0,"",""],["Mesa-Optimizers and Over-optimization Failure (Optimizing and Goodhart Effects, Clarifying Thoughts - Part 4)","Davidmanheim","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/89YgRc3NevJEDEen5/mesa-optimizers-and-over-optimization-failure-optimizing-and",0,"","goodharts-law"],["Attention in value-based choice as optimal sequential sampling.","Frederick Callaway and Tom Griffiths","2019","report","psyarxiv.com","psyarxiv.com/57v6k/",0,"",""],["Bayesian Relational Memory for Semantic Visual Navigation.","IEEE Transactions on Robotics","2019","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/iccv19-brm.pdf",0,"",""],["Combining reward information from multiple sources.","Dmitrii Krasheninnikov and 2 others","2019","report","rohinshah.com","rohinshah.com/wp-content/uploads/2019/12/Reward_Combination_NeurIPS_2019_Workshop_Camera_Ready.pdf",0,"",""],["Deception in finitely repeated security games.","Thanh H and 5 others","2019","report","strategicreasoning.org","strategicreasoning.org/wp-content/uploads/2018/12/Deception_in_Finitely_Repeated_Security_Games.pdf",0,"","deception"],["Demonstrating the Impact of Prior Knowledge in Risky Choice.","Mathew Hardy and Tom Griffiths","2019","report","psyarxiv.com","psyarxiv.com/jgxra",0,"",""],["Doing more with less: meta-reasoning and meta-learning in humans and machines.","Thomas L and 8 others","2019","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S2352154618302122",0,"",""],["Human-robot interaction for truck platooning using hierarchical dynamic games.","Elis Stefansson and 7 others","2019","report","iliad.stanford.edu","iliad.stanford.edu/pdfs/publications/stefansson2019human.pdf",0,"",""],["Identifying category representations for complex stimuli using discrete Markov chain Monte Carlo with people.","Anne S and 7 others","2019","report","link.springer.com","link.springer.com/article/10.3758/s13428-019-01201-9",0,"",""],["Learning Causal Trees with Latent Variables via Controlled Experimentation.","Prasad Tadepall and 3 others","2019","report","why19.causalai.net","why19.causalai.net/papers/SSS19_Paper_Upload_198.pdf",0,"",""],["Robust multi-agent reinforcement learning via minimax deep deterministic policy gradient.","Shihui Li and 5 others","2019","report","aima.eecs.berkeley.edu","aima.eecs.berkeley.edu/~russell/papers/aaai19-marl.pdf",0,"","agents policy"],["Scaling Out-of-Distribution Detection for Real-World Settings.","Dan Hendrycks and 5 others","2019","report","arxiv-export-lb.library.cornell.edu","arxiv-export-lb.library.cornell.edu/pdf/1911.11132",0,"",""],["The truth behind the myth of the folk theorem.","Joseph Y and 3 others","2019","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0899825619300582",0,"",""],["The Value of Abstraction.","Mark K and 5 others","2019","report","psyarxiv.com","psyarxiv.com/6fm9a/download?format=pdf",0,"",""],["Using Pre-Training Can Improve Model Robustness and Uncertainty.","Dan Hendrycks and 2 others","2019","report","proceedings.mlr.press","proceedings.mlr.press/v97/hendrycks19a/hendrycks19a.pdf",0,"","robustness"],["Why Can’t You Do That, HAL? Explaining Unsolvability of Planning Tasks.","Sarath Sreedharan and 3 others","2019","report","aair-lab.github.io","aair-lab.github.io/Publications/ijcai19.pdf",0,"",""],["Behaviour Suite for Reinforcement Learning","Ian Osband and 13 others","2019","paper","arXiv preprint","arxiv.org/abs/1908.03568",0,"","evals agents"],["AI Forecasting Dictionary (Forecasting infrastructure, part 1)","jacobjacob and goldhaber","2019","blog","EA Forum","forum.effectivealtruism.org/posts/gBL3yX4fAszePCnN2/ai-forecasting-dictionary-forecasting-infrastructure-part-1",0,"","forecasting"],["Four Ways An Impact Measure Could Help Alignment","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wJK944YqvFwjdbqCP/four-ways-an-impact-measure-could-help-alignment",0,"",""],["That which we call private","Úlfar Erlingsson and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1908.03566",0,"",""],["Verification and Transparency","DanielFilan","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/n3YRDJYCnQcDAw29G/verification-and-transparency",0,"","interpretability"],["Which of these five AI alignment research projects ideas are no good?","rmoehn","2019","blog","LessWrong","www.lesswrong.com/posts/m84D5GZMMERvH53q3/which-of-these-five-ai-alignment-research-projects-ideas-are",0,"",""],["August 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/08/06/august-2019-newsletter/",0,"",""],["Self-Supervised Learning and AGI Safety","Steven Byrnes","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/EMZeJ7vpfeF4GrWwm/self-supervised-learning-and-agi-safety",0,"",""],["Understanding Recent Impact Measures","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pf48kg9xCxJAcHmQc/understanding-recent-impact-measures",0,"",""],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features'","Logan Engstrom and 21 others","2019","report","Distill","distill.pub/2019/advex-bugs-discussion",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Adversarial Example Researchers Need to Expand What is Meant by 'Robustness'","Justin Gilmer and Dan Hendrycks","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-1",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Adversarial Examples are Just Bugs, Too","Preetum Nakkiran","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-5",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Adversarially Robust Neural Style Transfer","Reiichiro Nakano","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-4",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Discussion and Author Responses","Logan Engstrom and 4 others","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/original-authors",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Learning from Incorrectly Labeled Data","Eric Wallace","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-6",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Robust Feature Leakage","Gabriel Goh","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-2",0,"","robustness"],["A Discussion of 'Adversarial Examples Are Not Bugs, They Are Features': Two Examples of Useful, Non-Robust Features","Gabriel Goh","2019","report","Distill","distill.pub/2019/advex-bugs-discussion/response-3",0,"","robustness"],["A Survey of Early Impact Measures","Matthew Barnett","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TPy4RJvzogqqupDKk/a-survey-of-early-impact-measures",0,"",""],["An overview of arguments for concern about automation","alexlintz","2019","blog","EA Forum","forum.effectivealtruism.org/posts/eo6cqiYztthg6dizb/an-overview-of-arguments-for-concern-about-automation",0,"","governance"],["New paper: Corrigibility with Utility Preservation","Koen.Holtman","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/3uHgw2uW6BtR74yhQ/new-paper-corrigibility-with-utility-preservation",0,"",""],["Project Proposal: Considerations for trading off capabilities and safety impacts of AI research","David Scott Krueger (formerly: capybaralet)","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/y5fYPAyKjWePCsq3Y/project-proposal-considerations-for-trading-off-capabilities",0,"","evals"],["[AN #61] AI policy and governance, from two people in the field","Rohin Shah","2019","blog","LessWrong","www.lesswrong.com/posts/ZtXMM78zTqBe3eduo/an-61-ai-policy-and-governance-from-two-people-in-the-field",0,"","governance policy"],["AI Alignment Open Thread August 2019","habryka","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/aPwNaiSLjYP4XXZQW/ai-alignment-open-thread-august-2019",0,"",""],["Improving Deep Reinforcement Learning in Minecraft with Action Advice","Spencer Frazier and Mark Riedl","2019","paper","arXiv preprint","arxiv.org/abs/1908.01007",0,"",""],["Practical consequences of impossibility of value learning","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cnjWN4mzmWzggRnCJ/practical-consequences-of-impossibility-of-value-learning",0,"",""],["Neural Simplex Architecture","Dung T. Phan and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1908.00528",0,"","assurance"],["Why Subagents?","johnswentworth","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/3xF66BNSC5caZuKyC/why-subagents",0,"","agents"],["Contest: $1,000 for good questions to ask to an Oracle AI","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cSzaxcmeYW6z7cgtc/contest-usd1-000-for-good-questions-to-ask-to-an-oracle-ai",0,"","robustness"],["Towards a Theory of Intentions for Human-Robot Collaboration","Rocio Gomez and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.13275",0,"","evals"],["An upper bound for the background rate of human extinction","Andrew E. Snyder-Beattie and 2 others","2019","report","nature.com","www.nature.com/articles/s41598-019-47540-7",0,"",""],["Applying Overoptimization to Selection vs. Control (Optimizing and Goodhart Effects - Clarifying Thoughts, Part 3)","Davidmanheim","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zdeYiQgwYRs2bEmCK/applying-overoptimization-to-selection-vs-control-optimizing",0,"","goodharts-law deception"],["What does Optimization Mean, Again? (Optimizing and Goodhart Effects - Clarifying Thoughts, Part 2)","Davidmanheim","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BEMvcaeixt3uEqyBk/what-does-optimization-mean-again-optimizing-and-goodhart",0,"","goodharts-law"],["Is BERT Really Robust? A Strong Baseline for Natural Language Attack on Text Classification and Entailment","Di Jin and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.11932",0,"","evals"],["Some post- words for the future","Tamsin Leake","2019","blog","carado.moe","carado.moe/post-words.html",0,"",""],["The Artificial Intentional Stance","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NvqGmLBCtvQxfMs9m/the-artificial-intentional-stance",0,"",""],["A Unified Bellman Optimality Principle Combining Reward Maximization and Empowerment","Felix Leibfried and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.12392",0,"","agents policy robustness"],["Ought: why it matters and ways to help","Paul_Christiano","2019","blog","EA Forum","forum.effectivealtruism.org/posts/bxgxyn2m8zvrcAqd9/ought-why-it-matters-and-ways-to-help",0,"",""],["On the purposes of decision theory research","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/JSjagTDGdz2y6nNE3/on-the-purposes-of-decision-theory-research",0,"","theory"],["Ought: why it matters and ways to help","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cpewqG3MjnKJpCr7E/ought-why-it-matters-and-ways-to-help",0,"","deception"],["Reducing malicious use of synthetic media research: Considerations and potential release practices for machine learning","Aviv Ovadya and Jess Whittlestone","2019","paper","arXiv preprint","arxiv.org/abs/1907.11274",0,"",""],["IR-VIC: Unsupervised Discovery of Sub-goals for Transfer in RL","Nirbhay Modhe and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.10580",0,"",""],["AI Safety Debate and Its Applications","VojtaKovarik","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/5Kv2qNfRyXXihNrx2/ai-safety-debate-and-its-applications",0,"",""],["[AN #60] A new AI challenge: Minecraft agents that assist human players in creative mode","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ftEZPtMfnTQtXSnKd/an-60-a-new-ai-challenge-minecraft-agents-that-assist-human",0,"","agents"],["[Link] Thiel on GCRs","Milan_Griffes","2019","blog","EA Forum","forum.effectivealtruism.org/posts/vRuZGTYYCssraBavN/link-thiel-on-gcrs",0,"",""],["A Conceptually Well-Founded Characterization of Iterated Admissibility Using an \"All I Know\" Operator","Joseph Y. Halpern and Rafael Pass","2019","paper","EPTCS 297, 2019, pp. 221-232","arxiv.org/abs/1907.09106",0,"","agents"],["A system of different layers of abstraction for artificial intelligence","Alexander Serb and Themistoklis Prodromakis","2019","paper","arXiv preprint","arxiv.org/abs/1907.10508",0,"",""],["Why Build an Assistant in Minecraft?","Arthur Szlam and 13 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.09273",0,"","evals agents"],["July 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/07/19/july-2019-newsletter/",0,"",""],["Delegative Reinforcement Learning: learning to avoid traps with a little help","Vanessa Kosoy","2019","paper","SafeML ICLR 2019 Workshop","arxiv.org/abs/1907.08461",0,"",""],["Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery","Kristian Hartikainen and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.08225",0,"","agents policy robustness"],["Thoughts on the 5-10 Problem","Tofly","2019","blog","LessWrong","www.lesswrong.com/posts/uNjcYeMXXsnopajMg/thoughts-on-the-5-10-problem",0,"","agents theory"],["Historic trends in land speed records","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/historic-trends-in-land-speed-records/",0,"",""],["Robust Multi-Agent Reinforcement Learning via Minimax Deep Deterministic Policy Gradient","Shihui Li and 5 others","2019","report","aaai.org","www.aaai.org/ojs/index.php/AAAI/article/view/4327",0,"","agents policy"],["An Inductive Synthesis Framework for Verifiable Reinforcement Learning","He Zhu and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.07273",0,"","interpretability assurance monitoring"],["Jeff Hawkins on neuromorphic AGI within 20 years","Steven Byrnes","2019","blog","LessWrong","www.lesswrong.com/posts/FoJSa8mgLPT83g9e8/jeff-hawkins-on-neuromorphic-agi-within-20-years",0,"","forecasting"],["How Europe might matter for AI governance","stefan.torges","2019","blog","EA Forum","forum.effectivealtruism.org/posts/vnnmNYwi7QbJPstsz/how-europe-might-matter-for-ai-governance",0,"","governance policy"],["Grounding Value Alignment with Ethical Principles","Tae Wan Kim and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.05447",0,"",""],["The AI Timelines Scam","jessicata","2019","blog","LessWrong","www.lesswrong.com/posts/KnQs55tjxWopCzKsk/the-ai-timelines-scam",0,"","forecasting"],["The AI Timelines Scam","Jessica Taylor","2019","report","unstableontology.com","unstableontology.com/2019/07/11/the-ai-timelines-scam/",0,"","forecasting"],["An Optimistic Perspective on Offline Reinforcement Learning","Rishabh Agarwal and 2 others","2019","paper","Proceedings of the 37th International Conference on Machine\n  Learning, PMLR 119:104-114, 2020","arxiv.org/abs/1907.04543",0,"","deception agents policy"],["The Role of Cooperation in Responsible AI Development","Amanda Askell and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.04534",0,"",""],["Better-than-Demonstrator Imitation Learning via Automatically-Ranked Demonstrations","Daniel S. Brown and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.03976",0,"","policy"],["[AN #59] How arguments for AI risk have changed over time","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/MMvbNuAis3SSphu7D/an-59-how-arguments-for-ai-risk-have-changed-over-time",0,"",""],["Religion as Goodhart","shminux","2019","blog","LessWrong","www.lesswrong.com/posts/BjThrfnArSDXgECmD/religion-as-goodhart",0,"","goodharts-law"],["Some Comments on Stuart Armstrong's \"Research Agenda v0.9\"","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/GHNokcgERpLJwJnLW/some-comments-on-stuart-armstrong-s-research-agenda-v0-9",0,"",""],["Musings on Cumulative Cultural Evolution and AI","calebo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/K686EFdXysfRBdob2/musings-on-cumulative-cultural-evolution-and-ai",0,"",""],["Learning biases and rewards simultaneously","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/xxnPxELC4jLKaFKqG/learning-biases-and-rewards-simultaneously",0,"","policy"],["Quantifying the pathways to life using assembly spaces","Stuart M. Marshall and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.04649",0,"","evals deception"],["Learning a Behavioral Repertoire from Demonstrations","Niels Justesen and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.03046",0,"","deception policy"],["On Inductive Biases in Deep Reinforcement Learning","Matteo Hessel and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.02908",0,"","agents"],["Adversarial Robustness through Local Linearization","Chongli Qin and 8 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.02610",0,"","robustness training-data"],["Large Scale Adversarial Representation Learning","Jeff Donahue and Karen Simonyan","2019","paper","arXiv preprint","arxiv.org/abs/1907.02544",0,"","evals"],["Integration of Imitation Learning using GAIL and Reinforcement Learning using Task-achievement Rewards via Probabilistic Graphical Model","Akira Kinose and Tadahiro Taniguchi","2019","paper","Advanced Robotics, 2020, 34:16, 1055-1067","arxiv.org/abs/1907.02140",0,"","agents"],["Dynamics-Aware Unsupervised Discovery of Skills","Archit Sharma and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.01657",0,"","robustness"],["Generalizing from a few environments in safety-critical reinforcement learning","Zachary Kenton and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.01475",0,"","agents"],["Re-introducing Selection vs Control for Optimization (Optimizing and Goodhart Effects - Clarifying Thoughts, Part 1)","Davidmanheim","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/2neeoZ7idRbZf4eNC/re-introducing-selection-vs-control-for-optimization",0,"","goodharts-law"],["An Increasingly Manipulative Newsfeed","Michaël Trazzi","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/EpdXLNXyL4EYLFwF8/an-increasingly-manipulative-newsfeed",0,"","evals deception robustness"],["CHAI Newsletter #2 2019","CHAI","2019","report","drive.google.com","drive.google.com/file/d/1A8XFCUHKechIzAhsdgDdOX-BLNCUIMWR/view?usp=sharing",0,"",""],["Detecting Spiky Corruption in Markov Decision Processes","Jason Mancuso and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.00452",0,"","agents policy"],["Aligning a toy model of optimization","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/H5gXpFtg93qDMZ6Xn/aligning-a-toy-model-of-optimization",0,"","scalable-oversight policy robustness"],["Artificial Intelligence Governance and Ethics: Global Perspectives","Angela Daly and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1907.03848",0,"","evals governance"],["Conceptual Problems with UDT and Policy Selection","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9sYzoRnmqmxZm4Whf/conceptual-problems-with-udt-and-policy-selection",0,"","agents policy theory"],["Self-confirming prophecies, and simplified Oracle designs","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wJ3AqNPM7W4nfY5Bk/self-confirming-prophecies-and-simplified-oracle-designs",0,"",""],["Embedded Agency: Not Just an AI Problem","johnswentworth","2019","blog","LessWrong","www.lesswrong.com/posts/kFb8L4omGMk2kMK3K/embedded-agency-not-just-an-ai-problem",0,"","theory"],["Norms for Beneficial A.I.: A Computational Analysis of the Societal Value Alignment Problem","Pedro Fernandes and 2 others","2019","paper","AI Communications, vol. 33, no. 3-6, pp. 155-171, 2020","arxiv.org/abs/1907.03843",0,"",""],["Towards Empathic Deep Q-Learning","Bart Bussmann and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.10918",0,"","agents"],["Universal Litmus Patterns: Revealing Backdoor Attacks in CNNs","Soheil Kolouri and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.10842",0,"","benchmarks"],["Reinforcement Learning with Competitive Ensembles of Information-Constrained Primitives","Anirudh Goyal and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.10667",0,"","agents policy"],["Research Agenda in reverse: what *would* a solution look like?","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TR3eqQ2fnfKWzxxHL/research-agenda-in-reverse-what-would-a-solution-look-like",0,"",""],["[AN #58] Mesa optimization: what it is, and why we should care","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/XWPJfgBymBbL3jdFd/an-58-mesa-optimization-what-it-is-and-why-we-should-care",0,"",""],["Confidence-aware motion prediction for real-time collision avoidance <sup>1</sup>","David Fridovich-Keil and 6 others","2019","report","journals.sagepub.com","journals.sagepub.com/doi/10.1177/0278364919859436",0,"",""],["Learning to Interactively Learn and Assist","Mark Woodward and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.10187",0,"","agents"],["Machine Learning Projects on IDA","Owain_Evans and 2 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y9xD78kufNsF7wL6f/machine-learning-projects-on-ida",0,"","scalable-oversight"],["An AGI with Time-Inconsistent Preferences","James D. Miller and Roman Yampolskiy","2019","paper","arXiv preprint","arxiv.org/abs/1906.10536",0,"","agents"],["On the Feasibility of Learning, Rather than Assuming, Human Biases for Reward Inference","Rohin Shah and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.09624",0,"","deception agents"],["\"The Bitter Lesson\", an article about compute vs human knowledge in AI","the gears to ascension","2019","blog","LessWrong","www.lesswrong.com/posts/x5BFqov8rt2duhMvH/the-bitter-lesson-an-article-about-compute-vs-human",0,"","forecasting"],["Categorizing Wireheading in Partially Embedded Agents","Arushi Majha and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.09136",0,"","reward-hacking specification-gaming agents"],["Modeling AGI Safety Frameworks with Causal Influence Diagrams","Ramana Kumar","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HE5DL6XeomYxFab74/modeling-agi-safety-frameworks-with-causal-influence-1",0,"",""],["Information security careers for GCR reduction","ClaireZabel and lukeprog","2019","blog","EA Forum","forum.effectivealtruism.org/posts/ZJiCfwTy5dC4CoxqA/information-security-careers-for-gcr-reduction",0,"","governance robustness"],["Modeling AGI Safety Frameworks with Causal Influence Diagrams","Tom Everitt and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.08663",0,"","rlhf deception agents"],["Unsupervised State Representation Learning in Atari","Ankesh Anand and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.08226",0,"","evals deception agents"],["XLNet: Generalized Autoregressive Pretraining for Language Understanding","Zhilin Yang and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.08237",0,"",""],["1hr talk: Intro to AGI safety","Steven Byrnes","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/DbZDdupuffc4Xgm7H/1hr-talk-intro-to-agi-safety",0,"",""],["ICLR Safe ML Workshop Report","Victoria Krakovna","2019","report","futureoflife.org","futureoflife.org/2019/06/18/iclr-safe-ml-workshop-report/",0,"",""],["ICLR Safe ML workshop report","Victoria Krakovna","2019","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2019/06/18/iclr-safe-ml-workshop-report/",0,"",""],["Research Agenda v0.9: Synthesising a human's preferences into a utility function","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/CSEdLLEkap2pubjof/research-agenda-v0-9-synthesising-a-human-s-preferences-into",0,"",""],["Goal-conditioned Imitation Learning","Yiming Ding and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.05838",0,"","evals agents"],["Stand-Alone Self-Attention in Vision Models","Prajit Ramachandran and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.05909",0,"",""],["The Al Does Not Hate You: Superintelligence, Rationality and the Race to Save the World","Tom Chivers","2019","report","goodreads.com","www.goodreads.com/en/book/show/44154569",0,"",""],["Let's talk about \"Convergent Rationality\"","David Scott Krueger (formerly: capybaralet)","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pLZ3bdeng4u5W8Yft/let-s-talk-about-convergent-rationality-1",0,"","instrumental-convergence"],["Weight Agnostic Neural Networks","ADAM GAIER Google Brain DAVID HA Google Brain June 12 2019 Download PDF NeurIPS 2019 Slides","2019","report","weightagnostic.github.io","weightagnostic.github.io/",0,"",""],["Weight Agnostic Neural Networks","Adam Gaier and David Ha","2019","paper","arXiv preprint","arxiv.org/abs/1906.04358",0,"","evals"],["A Survey of Reinforcement Learning Informed by Natural Language","Jelena Luketina and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.03926",0,"",""],["E-LPIPS: Robust Perceptual Image Similarity via Random Transformation Ensembles","Markus Kettunen and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.03973",0,"","robustness"],["Self-Supervised Exploration via Disagreement","Deepak Pathak and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.04161",0,"","benchmarks agents policy"],["Tackling Climate Change with Machine Learning","David Rolnick and 21 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.05433",0,"",""],["Complexity no Bar to AI","Gwern Branwen","2019","blog","gwern.net","www.gwern.net/complexity.page",0,"",""],["AGI will drastically increase economies of scale","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Sn5NiiD5WBi4dLzaB/agi-will-drastically-increase-economies-of-scale",0,"","forecasting"],["For the past, in some ways only, we are moral degenerates","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SDr45pcgJJyvTqmZa/for-the-past-in-some-ways-only-we-are-moral-degenerates",0,"",""],["Likelihood Ratios for Out-of-Distribution Detection","Jie Ren and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.02845",0,"","benchmarks robustness training-data"],["New paper: “Risks from learned optimization”","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/06/07/new-paper-learned-optimization/",0,"",""],["Planning With Uncertain Specifications (PUnS)","Ankit Shah and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.03218",0,"","robustness"],["Risks from Learned Optimization: Conclusion and Related Work","evhub and 4 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4XPa3xa44jAWiCkmy/risks-from-learned-optimization-conclusion-and-related-work",0,"",""],["Risks from Learned Optimization: Conclusion and Related Work","evhub and 4 others","2019","blog","LessWrong","www.lesswrong.com/posts/4XPa3xa44jAWiCkmy/risks-from-learned-optimization-conclusion-and-related-work",0,"",""],["An Extensible Interactive Interface for Agent Design","Matthew Rahtz and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.02641",0,"","agents policy"],["Can You Trust Your Model's Uncertainty? Evaluating Predictive Uncertainty Under Dataset Shift","Yaniv Ovadia and 8 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.02530",0,"","evals benchmarks"],["Image Synthesis with a Single (Robust) Classifier","Shibani Santurkar and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.09453",0,"","robustness"],["Visualizing and Measuring the Geometry of BERT","Andy Coenen and Emily Reif","2019","paper","arXiv preprint","arxiv.org/abs/1906.02715",0,"",""],["[AN #57] Why we should focus on robustness in AI safety, and the analogous problems in programming","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZmvJ6kJ4ADcHcypYJ/an-57-why-we-should-focus-on-robustness-in-ai-safety-and-the",0,"","robustness"],["Deceptive Alignment","evhub and 4 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zthDPAjh9w6Ytbeks/deceptive-alignment",0,"","alignment-faking deception instrumental-convergence"],["Deceptive Alignment","evhub and 4 others","2019","blog","LessWrong","www.lesswrong.com/posts/zthDPAjh9w6Ytbeks/deceptive-alignment",0,"","alignment-faking deception instrumental-convergence"],["Methodology for discontinuous progress investigation","Asya Bergal","2019","blog","aiimpacts.org","aiimpacts.org/methodology-for-discontinuity-investigation/",0,"",""],["Teaching AI to Explain its Decisions Using Embeddings and Multi-Task Learning","Noel C. F. Codella and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.02299",0,"","evals training-data"],["The Inner Alignment Problem","evhub and 4 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pL56xPoniLvtMDQ4J/the-inner-alignment-problem",0,"",""],["The Inner Alignment Problem","evhub and 4 others","2019","blog","LessWrong","www.lesswrong.com/posts/pL56xPoniLvtMDQ4J/the-inner-alignment-problem",0,"",""],["Adversarial Robustness as a Prior for Learned Representations","Logan Engstrom and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.00945",0,"","robustness"],["Does Bayes Beat Goodhart?","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/YJq6R9Wgk5Atjx54D/does-bayes-beat-goodhart",0,"","goodharts-law agents"],["June 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/06/01/june-2019-newsletter/",0,"",""],["Learner-aware Teaching: Inverse Reinforcement Learning with Preferences and Constraints","Sebastian Tschiatschek and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1906.00429",0,"","agents policy"],["Moral Mazes and Short Termism","Zvi","2019","blog","LessWrong","www.lesswrong.com/posts/2Zsuv5uPFPNTACwzg/moral-mazes-and-short-termism",0,"","goodharts-law"],["Selection vs Control","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZDZmopKquzHYPRNxq/selection-vs-control",0,"","evals"],["The Principle of Unchanged Optimality in Reinforcement Learning Generalization","Alex Irpan and Xingyou Song","2019","paper","arXiv preprint","arxiv.org/abs/1906.00336",0,"","benchmarks agents policy robustness"],["2018 in review","Malo Bourgon","2019","blog","intelligence.org","intelligence.org/2019/05/31/2018-in-review/",0,"",""],["AI Governance and the Policymaking Process: Key Considerations for Reducing AI Risk","Brandon Perry and Risto Uuk","2019","report","mdpi.com","www.mdpi.com/2504-2289/3/2/26",0,"","governance policy"],["Conditions for Mesa-Optimization","evhub and 4 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/q2rCMHNXazALgQpGH/conditions-for-mesa-optimization",0,"",""],["Conditions for Mesa-Optimization","evhub and 4 others","2019","blog","LessWrong","www.lesswrong.com/posts/q2rCMHNXazALgQpGH/conditions-for-mesa-optimization",0,"",""],["Multiparty Dynamics and Failure Modes for Machine Learning and Artificial Intelligence","David Manheim","2019","report","mdpi.com","www.mdpi.com/2504-2289/3/2/21",0,"",""],["Risks from Learned Optimization: Introduction","evhub and 4 others","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/FkgsxrGf3QxhfLWHG/risks-from-learned-optimization-introduction",0,"",""],["Risks from Learned Optimization: Introduction","evhub and 4 others","2019","blog","LessWrong","www.lesswrong.com/posts/FkgsxrGf3QxhfLWHG/risks-from-learned-optimization-introduction",0,"",""],["Better Future through AI: Avoiding Pitfalls and Guiding AI Towards its Full Potential","Risto Miikkulainen and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.13178",0,"",""],["Imitation Learning as $f$-Divergence Minimization","Liyiming Ke and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.12888",0,"","deception policy"],["Asymptotically Unambitious Artificial General Intelligence","Michael K Cohen and 2 others","2019","paper","Proc.AAAI. 34 (2020) 2467-2476","arxiv.org/abs/1905.12186",0,"","instrumental-convergence"],["Defending Against Neural Fake News","Rowan Zellers and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.12616",0,"","deception training-data"],["Learning Representations by Humans, for Humans","Sophie Hilgard and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.12686",0,"","interpretability"],["SATNet: Bridging deep learning and logical reasoning using a differentiable satisfiability solver","Po-Wei Wang and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.12149",0,"",""],["A shift in arguments for AI risk","Richard_Ngo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hubbRt4DPegiA5gRR/a-shift-in-arguments-for-ai-risk",0,"","agents robustness"],["Causal Confusion in Imitation Learning","Pim de Haan and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.11979",0,"","agents policy robustness"],["AI-GAs: AI-generating algorithms, an alternate paradigm for producing general artificial intelligence","Jeff Clune","2019","paper","arXiv preprint","arxiv.org/abs/1905.10985",0,"","agents robustness"],["Cold Case: The Lost MNIST Digits","Chhavi Yadav and Léon Bottou","2019","paper","arXiv preprint","arxiv.org/abs/1905.10498",0,"",""],["On modelling the emergence of logical thinking","Cristian Ivan and Bipin Indurkhya","2019","paper","arXiv preprint","arxiv.org/abs/1905.09730",0,"","deception"],["AI-CARGO: A Data-Driven Air-Cargo Revenue Management System","Stefano Giovanni Rizzo and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.09130",0,"",""],["And the AI would have got away with it too, if...","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/92J4zJHkqmXTduxzY/and-the-ai-would-have-got-away-with-it-too-if",0,"",""],["Cognitive Model Priors for Predicting Human Decisions","David D. Bourgin and 4 others","2019","paper","Proceedings of the 36th International Conference on Machine\n  Learning, PMLR 97:5133-5141, 2019","arxiv.org/abs/1905.09397",0,"","benchmarks"],["Imitation Learning from Video by Leveraging Proprioception","Faraz Torabi and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.09335",0,"","agents policy"],["Where are people thinking and talking about global coordination for AI safety?","Wei Dai","2019","blog","LessWrong","www.lesswrong.com/posts/sM2sANArtSJE6duZZ/where-are-people-thinking-and-talking-about-global",0,"","governance"],["[AN #56] Should ML researchers stop running experiments before making hypotheses?","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/3zkXPo4ZTrDFZz7Sd/an-56-should-ml-researchers-stop-running-experiments-before",0,"",""],["By default, avoid ambiguous distant situations","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/PX8BB7Rqw7HedrSJd/by-default-avoid-ambiguous-distant-situations",0,"",""],["Perceptual Values from Observation","Ashley D. Edwards and Charles L. Isbell","2019","paper","arXiv preprint","arxiv.org/abs/1905.07861",0,"",""],["On Variational Bounds of Mutual Information","Ben Poole and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.06922",0,"",""],["From What to How: An Initial Review of Publicly Available AI Ethics Tools, Methods and Research to Translate Principles into Practices","Jessica Morley and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.06876",0,"",""],["Jade Leung: Why companies should be leading on AI governance","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/fniRhiPYw8b6FETsn/jade-leung-why-companies-should-be-leading-on-ai-governance",0,"","governance"],["The Algonauts Project: A Platform for Communication between the Sciences of Biological and Artificial Intelligence","Radoslaw Martin Cichy and 8 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.05675",0,"","benchmarks"],["Lie on the Fly: Strategic Voting in an Iterative Preference Elicitation Process","Lihi Dery and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.04933",0,"",""],["\"UDT2\" and \"against UD+ASSA\"","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/zd2DrbHApWypJD2Rz/udt2-and-against-ud-assa",0,"","theory"],["Coherent decisions imply consistent utilities","Eliezer Yudkowsky","2019","blog","LessWrong","www.lesswrong.com/posts/RQpNHSiWaXTvDxt6R/coherent-decisions-imply-consistent-utilities",0,"","theory"],["Complex Behavior from Simple (Sub)Agents","moridinamael","2019","blog","LessWrong","www.lesswrong.com/posts/3pKXC62C98EgCeZc4/complex-behavior-from-simple-sub-agents",0,"","agents"],["Integrating Artificial Intelligence into Weapon Systems","Philip Feldman and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.03899",0,"","governance robustness"],["May 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/05/10/may-2019-newsletter/",0,"",""],["Training human models is an unsolved problem","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/upP8PYgHfXgvgh3FF/training-human-models-is-an-unsolved-problem",0,"",""],["Aligning Recommender Systems as Cause Area","IvanVendrov","2019","blog","EA Forum","forum.effectivealtruism.org/posts/xzjQvqDYahigHcwgQ/aligning-recommender-systems-as-cause-area",0,"",""],["Meta-learning of Sequential Strategies","Pedro A. Ortega and 23 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.03030",0,"","agents policy"],["Toybox: A Suite of Environments for Experimental Evaluation of Deep Reinforcement Learning","Emma Tosch and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.02825",0,"","evals agents"],["Adversarial Examples Are Not Bugs, They Are Features","Andrew Ilyas and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.02175",0,"","robustness"],["Deconstructing Lottery Tickets: Zeros, Signs, and the Supermask","Hattie Zhou","2019","report","eng.uber.com","eng.uber.com/deconstructing-lottery-tickets/",0,"","deception"],["Value learning for moral essentialists","Charlie Steiner","2019","blog","LessWrong","www.lesswrong.com/posts/gPPduz7pTJHotuut6/value-learning-for-moral-essentialists",0,"",""],["[AN #55] Regulatory markets and international standards as a means of ensuring beneficial AI","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZQJ9H9ZeRF8mjB2aF/an-55-regulatory-markets-and-international-standards-as-a",0,"",""],["Deconstructing Lottery Tickets: Zeros, Signs, and the Supermask","Hattie Zhou and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.01067",0,"",""],["Meta-learners' learning dynamics are unlike learners'","Neil C. Rabinowitz","2019","paper","arXiv preprint","arxiv.org/abs/1905.01320",0,"",""],["Oracles, sequence predictors, and self-confirming predictions","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/i2dNFgbjnqZBfeitT/oracles-sequence-predictors-and-self-confirming-predictions",0,"",""],["PRECOG: PREdiction Conditioned On Goals in Visual Multi-Agent Settings","Nicholas Rhinehart and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.01296",0,"","agents"],["Self-confirming predictions can be arbitrarily bad","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/KoEY9CjrKe93ErYhd/self-confirming-predictions-can-be-arbitrarily-bad",0,"",""],["Transfer of Adversarial Robustness Between Perturbation Types","Daniel Kang and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1905.01034",0,"","evals robustness"],["Bridging Hamilton-Jacobi Safety Analysis and Reinforcement Learning","Jaime F. Fisac and 4 others","2019","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/8794107/",0,"",""],["Nash equilibriums can be arbitrarily bad","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jz5QoizH8HkQwWZ9Q/nash-equilibriums-can-be-arbitrarily-bad",0,"",""],["The relationship between Biological and Artificial Intelligence","George Cevora","2019","paper","arXiv preprint","arxiv.org/abs/1905.00547",0,"","evals deception"],["Challenges of Real-World Reinforcement Learning","Gabriel Dulac-Arnold and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.12901",0,"","policy"],["[AN #54] Boxing a finite-horizon AI system to keep it unambitious","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/AXCX2S4NbcyEu7hbX/an-54-boxing-a-finite-horizon-ai-system-to-keep-it",0,"",""],["What are some good examples of incorrigibility?","RyanCarey","2019","blog","LessWrong","www.lesswrong.com/posts/edi9Y4vYtdNRbui3u/what-are-some-good-examples-of-incorrigibility",0,"","robustness"],["Regulating AI: do we need new tools?","Otello Ardovino and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.12134",0,"","interpretability"],["Knowing When to Stop: Evaluation and Verification of Conformity to Output-size Specifications","Chenglong Wang and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.12004",0,"","evals robustness"],["Using Sub-Optimal Plan Detection to Identify Commitment Abandonment in Discrete Environments","Ramon Fraga Pereira and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.11737",0,"","agents"],["Ray Interference: a Source of Plateaus in Deep Reinforcement Learning","Tom Schaul and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.11455",0,"","policy"],["Strategic implications of AIs' ability to coordinate at low cost, for example by merging","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/gYaKZeBbSL4y2RLP3/strategic-implications-of-ais-ability-to-coordinate-at-low",0,"","agents forecasting"],["New paper: “Delegative reinforcement learning”","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/04/24/delegative-reinforcement-learning/",0,"","agents policy"],["Long-Term Future Fund: April 2019 grant recommendations","Habryka","2019","blog","EA Forum","forum.effectivealtruism.org/posts/CJJDwgyqT4gXktq6g/long-term-future-fund-april-2019-grant-recommendations",0,"","forecasting"],["Risk Structures: Towards Engineering Risk-aware Autonomous Systems","Mario Gleirscher","2019","paper","arXiv preprint","arxiv.org/abs/1904.10386",0,"","agents monitoring"],["AI Alignment Problem: “Human Values” don’t Actually Exist","avturchin","2019","blog","LessWrong","www.lesswrong.com/posts/ngqvnWGsvTEiTASih/ai-alignment-problem-human-values-don-t-actually-exist",0,"",""],["April 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/04/21/april-2019-newsletter/",0,"",""],["Optimization and Abstraction: A Synergistic Approach for Analyzing Neural Network Robustness","Greg Anderson and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.09959",0,"","evals benchmarks policy robustness"],["The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors","William H. Guss and 11 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.10079",0,"","agents"],["Any rebuttals of Christiano and AI Impacts on takeoff speeds?","SoerenMind","2019","blog","LessWrong","www.lesswrong.com/posts/PzAnWgqvfESgQEvdg/any-rebuttals-of-christiano-and-ai-impacts-on-takeoff-speeds",0,"","forecasting"],["Generative Exploration and Exploitation","Jiechuan Jiang and Zongqing Lu","2019","paper","arXiv preprint","arxiv.org/abs/1904.09605",0,"","agents policy"],["Helen Toner on China, CSET, and AI","Rob Bensinger","2019","blog","LessWrong","www.lesswrong.com/posts/cHwCBTwWiTdsqXNyn/helen-toner-on-china-cset-and-ai",0,"","governance"],["Alignment Newsletter #53","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jdMjzFf7tmT6ofLk9/alignment-newsletter-53",0,"",""],["When is a Prediction Knowledge?","Alex Kearney and Patrick M. Pilarski","2019","paper","arXiv preprint","arxiv.org/abs/1904.09024",0,"","evals deception agents"],["Analysing Neural Network Topologies: a Game Theoretic Approach","Julian Stier and 3 others","2019","paper","Procedia Computer Science 126 (2018): 234-243","arxiv.org/abs/1904.08166",0,"","deception robustness"],["Supporting global coordination in AI development: Why and how to contribute to international AI standards","pcihon","2019","blog","EA Forum","forum.effectivealtruism.org/posts/r5ZaEPbxHnM3cc5b8/supporting-global-coordination-in-ai-development-why-and-how",0,"","governance"],["Counterfactual Visual Explanations","Yash Goyal and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.07451",0,"","interpretability"],["End-to-End Robotic Reinforcement Learning without Reward Engineering","Avi Singh and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.07854",0,"","evals"],["HARK Side of Deep Learning -- From Grad Student Descent to Automated Machine Learning","Oguzhan Gencoglu and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.07633",0,"","robustness"],["Predicting human decisions with behavioral theories and machine learning","Ori Plonsky and 11 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.06866",0,"",""],["Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","Daniel S. Brown and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.06387",0,"","agents policy"],["Corrigibility as Constrained Optimisation","Henrik Åslund","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/cGLgs3t9md7v7cCm4/corrigibility-as-constrained-optimisation",0,"",""],["Alignment Newsletter One Year Retrospective","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/3onCb5ph3ywLQZMX2/alignment-newsletter-one-year-retrospective",0,"",""],["Alignment Newsletter One Year Retrospective","Rohin Shah","2019","blog","EA Forum","forum.effectivealtruism.org/posts/Prxqvhr9JFj7JyJRX/alignment-newsletter-one-year-retrospective",0,"",""],["Best reasons for pessimism about impact of impact measures?","TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/kCY9dYGLoThC3aG7w/best-reasons-for-pessimism-about-impact-of-impact-measures",0,"",""],["Semantics: Primes and Universals, a book review • carado.moe","Tamsin Leake","2019","blog","carado.moe","carado.moe/spu-review.html",0,"",""],["Open Questions about Generative Adversarial Networks","Augustus Odena","2019","report","Distill","distill.pub/2019/gan-open-problems",0,"",""],["Extending planning knowledge using ontologies for goal opportunities","Mohannad Babli and 2 others","2019","paper","31st IBIMA Conference (2018), INNOVATION MANAGEMENT AND EDUCATION\n  EXCELLENCE THROUGH VISION 2020, VOLS IV-VI (3199-3208)","arxiv.org/abs/1904.03606",0,"",""],["Reinforcement learning with imperceptible rewards","Vanessa Kosoy","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/aAzApjEpdYwAxnsAS/reinforcement-learning-with-imperceptible-rewards",0,"","deception agents"],["Alignment Newsletter #52","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/iWj7Ti9GA98M5JaMy/alignment-newsletter-52",0,"",""],["Alignment Newsletter #51","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/joxveLr8egM85YTy2/alignment-newsletter-51",0,"",""],["Defeating Goodhart and the \"closest unblocked strategy\" problem","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/PADPJ3xac5ogjEGwA/defeating-goodhart-and-the-closest-unblocked-strategy",0,"","goodharts-law"],["On AI and Compute","johncrox","2019","blog","LessWrong","www.lesswrong.com/posts/7MsKHa55HxGKFCN6z/on-ai-and-compute",0,"","forecasting"],["A Visual Exploration of Gaussian Processes","Jochen Görtler and 2 others","2019","report","Distill","distill.pub/2019/visual-exploration-gaussian-processes",0,"",""],["Are Query-Based Ontology Debuggers Really Helping Knowledge Engineers?","Patrick Rodler and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.01484",0,"","evals"],["Finding and Visualizing Weaknesses of Deep Reinforcement Learning Agents","Christian Rupprecht and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.01318",0,"","agents"],["CHAI Newsletter #1 2019","CHAI","2019","report","drive.google.com","drive.google.com/file/d/1gRVDU0KI3YNS6ObJ3yDEV8inXO82wLPC/view?usp=sharing",0,"",""],["Multitask Soft Option Learning","Maximilian Igl and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1904.01033",0,"","policy"],["New grants from the Open Philanthropy Project and BERI","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/04/01/new-grants-open-phil-beri/",0,"",""],["Informed Machine Learning -- A Taxonomy and Survey of Integrating Knowledge into Learning Systems","Laura von Rueden and 13 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.12394",0,"","evals training-data"],["Parfit's Escape (Filk)","Gordon Seidoh Worley","2019","blog","LessWrong","www.lesswrong.com/posts/wT9Ha4uNchdDoWTGg/parfit-s-escape-filk",0,"","theory"],["Alignment Newsletter #50","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/8NPx9FH2Zbv9gd9rX/alignment-newsletter-50",0,"",""],["Gradient Descent with Early Stopping is Provably Robust to Label Noise for Overparameterized Neural Networks","Mingchen Li and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.11680",0,"","robustness training-data"],["Historic trends in particle accelerator performance","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/particle-accelerator-performance-progress/",0,"",""],["A Concrete Proposal for Adversarial IDA","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jYvm4mmjvGHcPXtGL/a-concrete-proposal-for-adversarial-ida",0,"","scalable-oversight rlhf interpretability agents"],["The Last Global Era","Tamsin Leake","2019","blog","carado.moe","carado.moe/global-era.html",0,"",""],["Designing Normative Theories for Ethical and Legal Reasoning: LogiKEy Framework, Methodology, and Tool Support","Christoph Benzmüller and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.10187",0,"","deception agents governance"],["The LogBarrier adversarial attack: making effective use of decision boundary information","Chris Finlay and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.10396",0,"",""],["Visualizing memorization in RNNs","Andreas Madsen","2019","report","Distill","distill.pub/2019/memorization-in-rnns",0,"",""],["Worst-case guarantees","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/training-robust-corrigibility-ce0e0a3b9b4d",0,"","interpretability agents policy robustness"],["Improving Safety in Reinforcement Learning Using Model-Based Architectures and Human Intervention","Bharat Prakash and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.09328",0,"","evals agents"],["The Game Theory of Blackmail","Linda Linsefors","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wm2rdS3sDY9M5kpWb/the-game-theory-of-blackmail",0,"",""],["The Main Sources of AI Risk?","Daniel Kokotajlo and Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WXvt8bxYnwBYpy9oT/the-main-sources-of-ai-risk",0,"",""],["Towards Characterizing Divergence in Deep Q-Learning","Joshua Achiam and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.08894",0,"",""],["Alignment Newsletter #49","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WXhTxsphZtbQvp4Qt/alignment-newsletter-49",0,"",""],["What's wrong with these analogies for understanding Informed Oversight and IDA?","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/LigbvLH9yKR5Zhd6y/what-s-wrong-with-these-analogies-for-understanding-informed",0,"","scalable-oversight"],["Semantic Image Synthesis with Spatially-Adaptive Normalization","Taesung Park and 3 others","2019","paper","CVPR 2019","arxiv.org/abs/1903.07291",0,"",""],["What failure looks like","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HBxe6wdjxK239zajf/what-failure-looks-like",0,"","policy forecasting robustness"],["Boeing 737 MAX MCAS as an agent corrigibility failure","shminux","2019","blog","LessWrong","www.lesswrong.com/posts/JYvw2jv4R5HphXEd7/boeing-737-max-mcas-as-an-agent-corrigibility-failure",0,"","agents"],["Comparison of decision theories (with a focus on logical-counterfactual decision theories)","riceissa","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/QPhY8Nb7gtT5wvoPH/comparison-of-decision-theories-with-a-focus-on-logical",0,"","theory"],["Algorithms for Verifying Deep Neural Networks","Changliu Liu and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.06758",0,"",""],["Eric Drexler: Paretotopian goal alignment","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/fg6RrvtSJ2kxe9Ens/eric-drexler-paretotopian-goal-alignment",0,"",""],["Humans aren't agents - what then for value learning?","Charlie Steiner","2019","blog","LessWrong","www.lesswrong.com/posts/DsEuRrsenZ6piGpE6/humans-aren-t-agents-what-then-for-value-learning",0,"","agents"],["March 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/03/14/march-2019-newsletter/",0,"",""],["Deep Reinforcement Learning with Feedback-based Exploration","Jan Scholten and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.06151",0,"","policy robustness"],["A theory of human values","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qezBTig6p6p5xtL6G/a-theory-of-human-values",0,"","deception"],["Question: MIRI Corrigbility Agenda","algon33","2019","blog","LessWrong","www.lesswrong.com/posts/BScxwSun3K2MgpoNz/question-miri-corrigbility-agenda",0,"",""],["Reframing superintelligence _ Eric Drexler _ EA Global - London 2018-by Centre for Effective Altruism-video_id MircoV5LKvg-date 20190314","Eric Drexler","2019","report","drive.google.com","drive.google.com/file/d/1nacDSRDmZxaLP4wfRk0o-YPz2ecDO3mD/view?usp=share_link",0,"",""],["Alignment Newsletter #48","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BJ4Ek5BaJEKei3Czf/alignment-newsletter-48",0,"",""],["Applications are open for the MIRI Summer Fellows Program!","Colm Ó Riain","2019","blog","intelligence.org","intelligence.org/2019/03/10/applications-are-open-for-msfp/",0,"",""],["Designing agent incentives to avoid side effects","Vika and TurnTrout","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/eax34WBLNmB4Gv6so/designing-agent-incentives-to-avoid-side-effects",0,"","agents"],["Example population ethics: ordered discounted utility","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Ee29dFnPhaeRmYdMy/example-population-ethics-ordered-discounted-utility",0,"",""],["A new field guide for MIRIx","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/03/09/a-new-field-guide-for-mirix/",0,"",""],["Alignment Research Field Guide","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/PqMT9zGrNsGJNfiFR/alignment-research-field-guide",0,"",""],["CSER Advice to EU High-Level Expert Group on AI","HaydnBelfield","2019","blog","EA Forum","forum.effectivealtruism.org/posts/RE8eRsMh43YRZvTWg/cser-advice-to-eu-high-level-expert-group-on-ai",0,"","governance"],["CSER and FHI advice to UN High-level Panel on Digital Cooperation","HaydnBelfield","2019","blog","EA Forum","forum.effectivealtruism.org/posts/whDMv4NjsMcPrLq2b/cser-and-fhi-advice-to-un-high-level-panel-on-digital",0,"","governance"],["Smoothmin and personal identity","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/MxLK2fvEuijAYgsc2/smoothmin-and-personal-identity",0,"",""],["AI conference attendance","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/ai-conference-attendance/",0,"",""],["Meta-Dataset: A Dataset of Datasets for Learning to Learn from Few Examples","Eleni Triantafillou and 10 others","2019","paper","International Conference on Learning Representations (2020)","arxiv.org/abs/1903.03096",0,"","evals benchmarks"],["Self-Tuning Networks: Bilevel Optimization of Hyperparameters using Structured Best-Response Functions","Matthew MacKay and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.03088",0,"",""],["SLIDE : In Defense of Smart Algorithms over Hardware Acceleration for Large-Scale Deep Learning Systems","Beidi Chen and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.03129",0,"","evals"],["Activation Atlas","Shan Carter and 4 others","2019","report","Distill","distill.pub/2019/activation-atlas",0,"",""],["Historical economic growth trends","Katja Grace","2019","blog","aiimpacts.org","aiimpacts.org/historical-growth-trends/",0,"",""],["Learning Exploration Policies for Navigation","Tao Chen and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.01959",0,"","agents policy"],["Learning Latent Plans from Play","Corey Lynch and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.01973",0,"","agents"],["Simplified preferences needed; simplified preferences sufficient","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/sEqu6jMgnHG2fvaoQ/simplified-preferences-needed-simplified-preferences",0,"",""],["Stabilizing the Lottery Ticket Hypothesis","Jonathan Frankle and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.01611",0,"",""],["Three ways that \"Sufficiently optimized agents appear coherent\" can be false","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4K52SS7fm9mp5rMdX/three-ways-that-sufficiently-optimized-agents-appear",0,"","agents"],["Using Natural Language for Reward Shaping in Reinforcement Learning","Prasoon Goyal and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.02020",0,"","agents"],["A Strongly Asymptotically Optimal Agent in General Environments","Michael K. Cohen and 2 others","2019","paper","Proc.IJCAI (2019) 2179-2186","arxiv.org/abs/1903.01021",0,"","agents policy"],["Alignment Newsletter #47","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Aipqop4XpqPeGpWNi/alignment-newsletter-47",0,"",""],["Amanda Askell: AI safety needs social scientists","EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/ZLbS2WrHJdPGf24xh/amanda-askell-ai-safety-needs-social-scientists",0,"",""],["Finding the variables","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pHHhyZX5zwvwNqDXm/finding-the-variables",0,"",""],["IRL 1/8: Inverse Reinforcement Learning and the problem of degeneracy","RAISE","2019","blog","LessWrong","www.lesswrong.com/posts/bAcJxQeKJcBBnE7bc/irl-1-8-inverse-reinforcement-learning-and-the-problem-of",0,"",""],["Model Primitive Hierarchical Lifelong Reinforcement Learning","Bohan Wu and 2 others","2019","paper","International Conference on Autonomous Agents and Multiagent\n  Systems (AAMAS 2019)","arxiv.org/abs/1903.01567",0,"","interpretability robustness"],["Using Causal Analysis to Learn Specifications from Task Demonstrations","Daniel Angelov and 2 others","2019","paper","Proceedings of the 18th International Conference on Autonomous\n  Agents and MultiAgent Systems, Pages 1341-1349, Montreal QC, Canada, May 13 -\n  17, 2019","arxiv.org/abs/1903.01267",0,"",""],["Hacking Google reCAPTCHA v3 using Reinforcement Learning","Ismail Akrout and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.01003",0,"","agents"],["Autocurricula and the Emergence of Innovation from Social Interaction: A Manifesto for Multi-Agent Intelligence Research","Joel Z. Leibo and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.00742",0,"","agents"],["Learning Robust Representations by Projecting Superficial Statistics Out","Haohan Wang and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.06256",0,"","evals robustness training-data"],["Primates vs birds: Is one brain architecture better than the other?","Tegan McCaslin","2019","blog","aiimpacts.org","aiimpacts.org/primates-vs-birds-is-one-brain-architecture-better-than-the-other/",0,"",""],["AI safety needs social scientists _ Amanda Askell _ EA Global - London 2018-by Centre for Effective Altruism-video_id TWHcK-BNo1w-date 20190301","Amanda Askell","2019","report","drive.google.com","drive.google.com/file/d/1hkDZvLBP3FsIx-qHzNbdrakdTq44VkgA/view?usp=share_link",0,"",""],["The Ethics of AI Ethics -- An Evaluation of Guidelines","Thilo Hagendorff","2019","paper","Minds & Machines, 2020","arxiv.org/abs/1903.03425",0,"","evals"],["Attention is not Explanation","Sarthak Jain and Byron C. Wallace","2019","paper","arXiv preprint","arxiv.org/abs/1902.10186",0,"","interpretability"],["Diagnosing Bottlenecks in Deep Q-learning Algorithms","Justin Fu and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.10250",0,"",""],["How to get value learning and reference wrong","Charlie Steiner","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NcahJFd5S5RNupxFT/how-to-get-value-learning-and-reference-wrong",0,"",""],["RAISE is launching their MVP","anonymous","2019","blog","LessWrong","www.lesswrong.com/posts/WgnAEXw5fXaW9p5PS/raise-is-launching-their-mvp",0,"","scalable-oversight"],["Understanding Agent Incentives using Causal Influence Diagrams. Part I: Single Action Settings","Tom Everitt and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.09980",0,"","agents"],["Challenges for an Ontology of Artificial Intelligence","Scott H. Hawley","2019","paper","arXiv preprint","arxiv.org/abs/1903.03171",0,"","evals agents"],["Embedded Agency","Abram Demski and Scott Garrabrant","2019","paper","arXiv preprint","arxiv.org/abs/1902.09469",0,"","agents robustness theory"],["February 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/02/25/february-2019-newsletter/",0,"",""],["Improving Robustness of Machine Translation with Synthetic Noise","Vaibhav Vaibhav and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.09508",0,"","robustness"],["Verification of Non-Linear Specifications for Neural Networks","Chongli Qin and 9 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.09592",0,"","evals training-data"],["Can HCH epistemically dominate Ramanujan?","zhukeepa","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4qY9zEHLa2su4PkQ4/can-hch-epistemically-dominate-ramanujan",0,"","scalable-oversight"],["Thoughts on Human Models","Guest","2019","blog","intelligence.org","intelligence.org/2019/02/22/thoughts-on-human-models/",0,"",""],["Alignment Newsletter #46","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NeJnFmeNAXACASX8P/alignment-newsletter-46",0,"",""],["What type of Master's is best for AI policy work?","Milan_Griffes","2019","blog","EA Forum","forum.effectivealtruism.org/posts/cjNkvXSFPxTuYBaaZ/what-type-of-master-s-is-best-for-ai-policy-work",0,"","governance policy"],["Confused about AI research as a means of addressing AI risk","Eli Rose","2019","blog","EA Forum","forum.effectivealtruism.org/posts/hnFkzrEuWfvRm6Ao6/confused-about-ai-research-as-a-means-of-addressing-ai-risk",0,"",""],["FHI Report: Stable Agreements in Turbulent Times","Cullen","2019","blog","EA Forum","forum.effectivealtruism.org/posts/3LfyB9Dpan7fx5Jk3/fhi-report-stable-agreements-in-turbulent-times",0,"","governance"],["Quantifying Perceptual Distortion of Adversarial Examples","Matt Jordan and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.08265",0,"","robustness"],["Thoughts on Human Models","Ramana Kumar and Scott Garrabrant","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BKjJJH2cRpJcAnP7T/thoughts-on-human-models",0,"","evals agents"],["From Language to Goals: Inverse Reinforcement Learning for Vision-Based Instruction Following","Justin Fu and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.07742",0,"","policy"],["Meta-Weight-Net: Learning an Explicit Mapping For Sample Weighting","Jun Shu and 6 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.07379",0,"","deception training-data"],["Pavlov Generalizes","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/XTgkhjNTEi97WHMi6/pavlov-generalizes",0,"","agents robustness theory"],["World Discovery Models","Mohammad Gheshlaghi Azar and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.07685",0,"","agents"],["AI Safety Needs Social Scientists","Geoffrey Irving and Amanda Askell","2019","report","Distill","distill.pub/2019/safety-needs-social-scientists",0,"",""],["Parenting: Safe Reinforcement Learning from Human Input","Christopher Frye and Ilya Feige","2019","paper","arXiv preprint","arxiv.org/abs/1902.06766",0,"","reward-hacking agents"],["Regularizing Black-box Models for Improved Interpretability","Gregory Plumb and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.06787",0,"","interpretability"],["STRIP: A Defence Against Trojan Attacks on Deep Neural Networks","Yansong Gao and 5 others","2019","paper","In 2019 Annual Computer Security Applications Conference (ACSAC\n  19), December 9-13, 2019, San Juan, PR, USA. ACM, New York, NY, USA","arxiv.org/abs/1902.06531",0,"","interpretability evals robustness training-data"],["Was ist eine Professur fuer Kuenstliche Intelligenz?","Kristian Kersting and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1903.09516",0,"",""],["Self-supervised Visual Feature Learning with Deep Neural Networks: A Survey","Longlong Jing and Yingli Tian","2019","paper","arXiv preprint","arxiv.org/abs/1902.06162",0,"","evals benchmarks"],["How does OpenAI's language model affect our AI timeline estimates?","jimrandomh","2019","blog","LessWrong","www.lesswrong.com/posts/WjxSFmm7GvWEMovzR/how-does-openai-s-language-model-affect-our-ai-timeline",0,"","forecasting"],["How the MtG Color Wheel Explains AI Safety","Scott Garrabrant","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9CKBtxWtjvminNTmC/how-the-mtg-color-wheel-explains-ai-safety",0,"",""],["Alignment Newsletter #45","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/psKxyNGH9HuxvpFPB/alignment-newsletter-45",0,"",""],["Fixed-point solutions to the regress problem in normative uncertainty","Philip Trammell","2019","report","doi.org","doi.org/10.1007/s11229-019-02098-9",0,"",""],["Unsupervised Visuomotor Control through Distributional Planning Networks","Tianhe Yu and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.05542",0,"",""],["Three Biases That Made Me Believe in AI Risk","beth​","2019","blog","EA Forum","forum.effectivealtruism.org/posts/Yseu9oG3gnb6ERc7n/three-biases-that-made-me-believe-in-ai-risk",0,"",""],["Deep Reinforcement Learning from Policy-Dependent Human Feedback","Dilip Arumugam and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.04257",0,"","rlhf agents policy"],["Learning preferences by looking at the world","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/7f6DNZhracD7RvxMr/learning-preferences-by-looking-at-the-world",0,"",""],["Nuances with ascription universality","evhub","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/R5Euq7gZgobJi5S25/nuances-with-ascription-universality",0,"","interpretability"],["Preferences Implicit in the State of the World","Rohin Shah and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.04198",0,"","evals agents"],["Coherent behaviour in the real world is an incoherent concept","Richard_Ngo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/vphFJzK3mWA4PJKAg/coherent-behaviour-in-the-real-world-is-an-incoherent",0,"","deception agents"],["Investigation into the relationship between neuron count and intelligence across differing cortical architectures","Tegan McCaslin","2019","blog","aiimpacts.org","aiimpacts.org/investigation-into-the-relationship-between-neuron-count-and-intelligence-across-differing-cortical-architectures/",0,"",""],["Learning Preferences by Looking at the World","Daniel Seita","2019","report","bair.berkeley.edu","bair.berkeley.edu/blog/2019/02/11/learning_preferences/",0,"",""],["Our 2018 Fundraiser Review","Colm Ó Riain","2019","blog","intelligence.org","intelligence.org/2019/02/11/our-2018-fundraiser-review/",0,"",""],["Would I think for ten thousand years?","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/6zwW9oGaHbuMuvnmX/would-i-think-for-ten-thousand-years",0,"",""],["Some Thoughts on Metaphilosophy","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/EByDsY9S3EDhhfFzC/some-thoughts-on-metaphilosophy",0,"",""],["The Argument from Philosophical Difficulty","Wei Dai","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/w6d7XBCegc96kz4n3/the-argument-from-philosophical-difficulty",0,"","robustness"],["Ben Garfinkel: How sure are we about this AI stuff?","bgarfinkel and EA Global","2019","blog","EA Forum","forum.effectivealtruism.org/posts/9sBAW3qKppnoG3QPq/ben-garfinkel-how-sure-are-we-about-this-ai-stuff",0,"",""],["HCH is not just Mechanical Turk","William_S","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4JuKoFguzuMrNn6Qr/hch-is-not-just-mechanical-turk",0,"","policy robustness"],["Reinforcement Learning in the Iterated Amplification Framework","William_S","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/fq7Ehb2oWwXtZic8S/reinforcement-learning-in-the-iterated-amplification",0,"","scalable-oversight"],["Ask Not What AI Can Do, But What AI Should Do: Towards a Framework of Task Delegability","Brian Lubars and Chenhao Tan","2019","paper","arXiv preprint","arxiv.org/abs/1902.03245",0,"","ai-control"],["Certified Adversarial Robustness via Randomized Smoothing","Jeremy M Cohen and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.02918",0,"","robustness"],["Evidence on good forecasting practices from the Good Judgment Project","Daniel Kokotajlo","2019","blog","aiimpacts.org","aiimpacts.org/evidence-on-good-forecasting-practices-from-the-good-judgment-project/",0,"","forecasting robustness"],["Evidence on good forecasting practices from the Good Judgment Project: an accompanying blog post","Daniel Kokotajlo","2019","blog","aiimpacts.org","aiimpacts.org/evidence-on-good-forecasting-practices-from-the-good-judgment-project-an-accompanying-blog-post/",0,"","forecasting robustness"],["Hybrid Models with Deep and Invertible Features","Eric Nalisnick and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.02767",0,"","robustness"],["Alignment Newsletter #44","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/54jDNkonygFheKL9H/alignment-newsletter-44",0,"",""],["Security amplification","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hjEaZgyQ2iprDhkg8/security-amplification",0,"","scalable-oversight deception agents"],["Test Cases for Impact Regularisation Methods","DanielFilan","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wzPzPmAsG3BwrBrwy/test-cases-for-impact-regularisation-methods",0,"",""],["Why companies should be leading on AI governance _ Jade Leung _ EA Global - London 2018-by Centre for Effective Altruism-video_id AVDIQvJVhso-date 20190207","Jade Leung","2019","report","drive.google.com","drive.google.com/file/d/1c9Ee4DjntG3BWZiLtaDALoLsKkuIa8YF/view?usp=share_link",0,"","governance"],["PUTWorkbench: Analysing Privacy in AI-intensive Systems","Saurabh Srivastava and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.01580",0,"",""],["When to use quantilization","RyanCarey","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Rs6vZCrnQFWQ4p37P/when-to-use-quantilization",0,"","goodharts-law"],["(notes on) Policy Desiderata for Superintelligent AI: A Vector Field Approach","Ben Pace","2019","blog","LessWrong","www.lesswrong.com/posts/hyfedqhgCQriBB9wT/notes-on-policy-desiderata-for-superintelligent-ai-a-vector",0,"","governance policy"],["Conclusion to the sequence on value learning","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TE5nJ882s5dCMkBB8/conclusion-to-the-sequence-on-value-learning",0,"",""],["Constructing Goodhart","johnswentworth","2019","blog","LessWrong","www.lesswrong.com/posts/NwaNPHYhXDc9LkK8J/constructing-goodhart",0,"","goodharts-law"],["How sure are we about this AI stuff _ Ben Garfinkel _ EA Global - London 2018-by Centre for Effective Altruism-video_id E8PGcoLDjVk-date 20190204","Ben Garfinkel","2019","report","drive.google.com","drive.google.com/file/d/1XWDpS-QTI2fmFZw4ugcmXmQqj2yuSEP4/view?usp=share_link",0,"",""],["How does Gradient Descent Interact with Goodhart?","Scott Garrabrant","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/pcomQ4Fwi7FnfBZBR/how-does-gradient-descent-interact-with-goodhart",0,"","goodharts-law"],["Drexler on AI Risk","PeterMcCluskey","2019","blog","LessWrong","www.lesswrong.com/posts/bXYtDfMTNbjCXFQPh/drexler-on-ai-risk",0,"",""],["January 2019 Newsletter","Rob Bensinger","2019","blog","intelligence.org","intelligence.org/2019/01/31/january-2019-newsletter/",0,"",""],["The Hanabi Challenge: A New Frontier for AI Research","Nolan Bard and 14 others","2019","paper","arXiv preprint","arxiv.org/abs/1902.00506",0,"","evals agents"],["Human-Centered Artificial Intelligence and Machine Learning","Mark O. Riedl","2019","paper","arXiv preprint","arxiv.org/abs/1901.11184",0,"","interpretability"],["Reliability amplification","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/6fMvGoyy3kgnonRNM/reliability-amplification",0,"","scalable-oversight agents"],["The case for building expertise to work on US AI policy, and how to do it","80000_Hours","2019","blog","EA Forum","forum.effectivealtruism.org/posts/oHiQcBtDJiqPLnoAE/the-case-for-building-expertise-to-work-on-us-ai-policy-and",0,"","governance policy"],["A Comparative Analysis of Expected and Distributional Reinforcement Learning","Clare Lyle and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.11084",0,"","agents policy"],["Deconfusing Logical Counterfactuals","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BRuWm4GxcTNPn4XDX/deconfusing-logical-counterfactuals",0,"","theory"],["Wireheading is in the eye of the beholder","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/BvctuKocyWR4YYea3/wireheading-is-in-the-eye-of-the-beholder",0,"","reward-hacking"],["Alignment Newsletter #43","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Dm2mXk94PATehdr9J/alignment-newsletter-43",0,"",""],["Can there be an indescribable hellworld?","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/rArsypGqq49bk4iRr/can-there-be-an-indescribable-hellworld",0,"",""],["How much can value learning be disentangled?","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/Q7WiHdSSShkNsgDpa/how-much-can-value-learning-be-disentangled",0,"","robustness"],["Which textbook would you recommend to learn decision theory?","supermartingale","2019","blog","LessWrong","www.lesswrong.com/posts/EpecJQg7oMcxpgS7L/which-textbook-would-you-recommend-to-learn-decision-theory",0,"","theory"],["Lyapunov-based Safe Policy Optimization for Continuous Control","Yinlam Chow and 4 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.10031",0,"","evals agents policy"],["Techniques for optimizing worst-case performance","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/E2aZ9Xwdz3i2ghPtn/techniques-for-optimizing-worst-case-performance",0,"","scalable-oversight interpretability agents training-data"],["Using Pre-Training Can Improve Model Robustness and Uncertainty","Dan Hendrycks and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.09960",0,"","evals robustness"],["Epistemic Therapy for Bias in Automated Decision-Making","Thomas Krendl Gilbert and Yonatan Mintz","2019","report","dl.acm.org","dl.acm.org/doi/10.1145/3306618.3314294",0,"",""],["Specifying AI Objectives As a Human-AI Collaboration Problem","Anca Dragan","2019","report","doi.acm.org","doi.acm.org/10.1145/3306618.3314227",0,"",""],["Future directions for narrow value learning","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/MxadmSXHnoCupsWqx/future-directions-for-narrow-value-learning",0,"",""],["Forecasting Transformative AI: An Expert Survey","Ross Gruetzemacher and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.08579",0,"","policy forecasting"],["Informed oversight","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/informed-oversight-18fcb5d3d1e1",0,"","agents robustness"],["Is Agent Simulates Predictor a \"fair\" problem?","Chris_Leong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HvLcGmr2APwqauFzW/is-agent-simulates-predictor-a-fair-problem",0,"","agents theory"],["Theoretically Principled Trade-off between Robustness and Accuracy","Hongyang Zhang and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.08573",0,"","deception robustness"],["Thoughts on reward engineering","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/NtX7LKhCXMW2vjWx6/thoughts-on-reward-engineering",0,"","scalable-oversight evals agents robustness"],["Thoughts on reward engineering","paulfchristiano","2019","blog","LessWrong","www.lesswrong.com/posts/NtX7LKhCXMW2vjWx6/thoughts-on-reward-engineering",0,"","scalable-oversight"],["Allowing a formal proof system to self improve while avoiding Lobian obstacles.","Donald Hobson","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/hchfRj4qa4hFZxhKM/allowing-a-formal-proof-system-to-self-improve-while",0,"",""],["Disentangling arguments for the importance of AI safety","richard_ngo","2019","blog","EA Forum","forum.effectivealtruism.org/posts/LprnaEj3uhkmYtmat/disentangling-arguments-for-the-importance-of-ai-safety",0,"",""],["Learning with catastrophes","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/qALeGJ9nPcs9eC9Af/learning-with-catastrophes",0,"","scalable-oversight agents robustness"],["S-Curves for Trend Forecasting","Matt Goldenberg","2019","blog","LessWrong","www.lesswrong.com/posts/oaqKjHbgsoqEXBMZ2/s-curves-for-trend-forecasting",0,"","forecasting"],["Alignment Newsletter #42","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/2dt8miopNAvhKZPNf/alignment-newsletter-42",0,"",""],["Disentangling arguments for the importance of AI safety","Richard_Ngo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/JbcWQCxKWn3y49bNB/disentangling-arguments-for-the-importance-of-ai-safety",0,"","goodharts-law agents robustness"],["Announcement: AI alignment prize round 4 winners","cousin_it","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/nDHbgjdddG5EN6ocg/announcement-ai-alignment-prize-round-4-winners",0,"",""],["Capability amplification","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/t3AJW5jP3sk36aGoC/capability-amplification-1",0,"","scalable-oversight"],["Following human norms","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/eBd6WvzhuqduCkYv3/following-human-norms",0,"","agents robustness"],["Reward uncertainty","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZiLLxaLB5CCofrzPp/reward-uncertainty",0,"",""],["Why not tool AI?","smithee","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/jkxkMTGfZDzBEaaY8/why-not-tool-ai",0,"",""],["Theory of Minds: Understanding Behavior in Groups Through Inverse Planning","Michael Shum and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.06085",0,"","deception agents"],["Alignment Newsletter #41","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/YnBFravZQ5qm6Nmyh/alignment-newsletter-41",0,"",""],["Amplifying the Imitation Effect for Reinforcement Learning of UCAV's Mission Execution","Gyeong Taek Lee and Chang Ouk Kim","2019","paper","arXiv preprint","arxiv.org/abs/1901.05856",0,"","agents policy"],["Anthropics is pretty normal","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/uAqs5Q3aGEen3nKeX/anthropics-is-pretty-normal",0,"",""],["Attentive Neural Processes","Hyunjik Kim and 7 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.05761",0,"",""],["Debate AI and the Decision to Release an AI","Chris_Leong","2019","blog","LessWrong","www.lesswrong.com/posts/8R9XcZKZ4f38aRJ9A/debate-ai-and-the-decision-to-release-an-ai",0,"",""],["The reward engineering problem","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4nZRzoGTqg8xy5rr8/the-reward-engineering-problem",0,"","scalable-oversight interpretability agents"],["The reward engineering problem","paulfchristiano","2019","blog","LessWrong","www.lesswrong.com/posts/4nZRzoGTqg8xy5rr8/the-reward-engineering-problem",0,"","scalable-oversight"],["What AI Safety Researchers Have Written About the Nature of Human Values","avturchin","2019","blog","LessWrong","www.lesswrong.com/posts/GermiEmcS6xuZ2gBh/what-ai-safety-researchers-have-written-about-the-nature-of",0,"",""],["Artificial Intelligence and Robotization","Martina Kunz and Seán Ó hÉigeartaigh","2019","report","papers.ssrn.com","papers.ssrn.com/abstract=3310421",0,"",""],["Human-AI Interaction","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/4783ufKpx8xvLMPc6/human-ai-interaction",0,"","governance robustness"],["Identifying and Correcting Label Bias in Machine Learning","Heinrich Jiang and Ofir Nachum","2019","paper","arXiv preprint","arxiv.org/abs/1901.04966",0,"","evals deception agents"],["CDT=EDT=UDT","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/WkPf6XCzfJLCm2pbK/cdt-edt-udt",0,"","theory"],["Directions and desiderata for AI alignment","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/kphJvksj5TndGapuh/directions-and-desiderata-for-ai-alignment",0,"","scalable-oversight"],["Dutch-Booking CDT","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/wkNQdYj47HX33noKv/dutch-booking-cdt",0,"","theory"],["Towards formalizing universality","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/M8WdeNWacMrmorNdd/towards-formalizing-universality",0,"",""],["When is CDT Dutch-Bookable?","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/TJT2oBMGaZTE7f2z2/when-is-cdt-dutch-bookable",0,"","theory"],["Ambitious vs. narrow value learning","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/SvuLhtREMy8wRBzpC/ambitious-vs-narrow-value-learning",0,"","deception instrumental-convergence robustness"],["Comments on CAIS","Richard_Ngo","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HvNAmkXPTSoA4dvzv/comments-on-cais",0,"","agents"],["Non-Consequentialist Cooperation?","abramdemski","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/F9vcbEMKW48j4Z6h9/non-consequentialist-cooperation",0,"",""],["Towards formalizing universality","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/towards-formalizing-universality-409ab893a456",0,"","debate interpretability agents"],["A New Tensioning Method using Deep Reinforcement Learning for Surgical Pattern Cutting","Thanh Thi Nguyen and 3 others","2019","paper","2019 IEEE International Conference on Industrial Technology (ICIT)","arxiv.org/abs/1901.03327",0,"","policy robustness"],["Universality and consequentialism within HCH","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/universality-and-consequentialism-within-hch-c0bee00365bd",0,"","scalable-oversight agents robustness"],["What is narrow value learning?","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/vX7KirQwHsBaSEdfK/what-is-narrow-value-learning",0,"","robustness"],["AlphaGo Zero and capability amplification","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/HA3oArypzNANvXC38/alphago-zero-and-capability-amplification",0,"","scalable-oversight agents policy"],["Making AI meaningful again","Jobst Landgrebe and Barry Smith","2019","paper","arXiv preprint","arxiv.org/abs/1901.02918",0,"","deception"],["No surjection onto function space for manifold X","Stuart_Armstrong","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/eqi83c2nNSX7TFSfW/no-surjection-onto-function-space-for-manifold-x",0,"",""],["Alignment Newsletter #40","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/kuL7YmsuQJ9v6xNhK/alignment-newsletter-40",0,"",""],["Reframing Superintelligence: Comprehensive AI Services as General Intelligence","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/x3fNwSe5aWZb5yXEG/reframing-superintelligence-comprehensive-ai-services-as",0,"",""],["Risk-Aware Active Inverse Reinforcement Learning","Daniel S. Brown and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.02161",0,"","policy"],["Robust Change Captioning","Dong Huk Park and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.02527",0,"","agents"],["AI safety without goal-directed behavior","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/tHxXdAn8Yuiy9y2pZ/ai-safety-without-goal-directed-behavior",0,"",""],["Paired Open-Ended Trailblazer (POET): Endlessly Generating Increasingly Complex and Diverse Learning Environments and Their Solutions","Rui Wang and 3 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.01753",0,"","agents"],["Failures of UDT-AIXI, Part 1: Improper Randomizing","Diffractor","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/mrZp6qC7DDXKQZeeC/failures-of-udt-aixi-part-1-improper-randomizing",0,"",""],["Supervising strong learners by amplifying weak experts","paulfchristiano","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/xKvzpodBGcPMq7TqE/supervising-strong-learners-by-amplifying-weak-experts",0,"","scalable-oversight evals"],["Hierarchical Reinforcement Learning via Advantage-Weighted Information Maximization","Takayuki Osa and 2 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.01365",0,"","policy"],["Will humans build goal-directed agents?","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/9zpT9dikrrebdq3Jf/will-humans-build-goal-directed-agents",0,"","interpretability agents robustness"],["Two More Decision Theory Problems for Humans","Wei Dai","2019","blog","LessWrong","www.lesswrong.com/posts/6RjL996E8Dsz3vHPk/two-more-decision-theory-problems-for-humans",0,"","evals theory"],["A Comprehensive Survey on Graph Neural Networks","Zonghan Wu and 5 others","2019","paper","arXiv preprint","arxiv.org/abs/1901.00596",0,"","evals benchmarks"],["Logical inductors in multistable situations.","Donald Hobson","2019","blog","LessWrong","www.lesswrong.com/posts/vQFiqbH6AfB7uYhQL/logical-inductors-in-multistable-situations",0,"","theory"],["Universality and security amplification","Paul Christiano","2019","report","ai-alignment.com","ai-alignment.com/universality-and-security-amplification-551b314a3bab",0,"","scalable-oversight"],["2018-19 New Year review","Victoria Krakovna","2019","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2019/01/01/2018-19-new-year-review/",0,"",""],["AI Insights Dataset Analysis","Colleen McKenzie and J Bryce Hidysmith","2019","report","mediangroup.org","mediangroup.org/docs/insights-analysis.pdf",0,"",""],["AI Safety Open Problems","Mati Roy","2019","report","docs.google.com","docs.google.com/document/d/1J2fOOF-NYiPC0-J3ZGEfE0OhA-QcOInhlvWjr1fAsS0/edit?usp=embed_facebook",0,"",""],["Alignment Newsletter #39","Rohin Shah","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/waAfXvcmbqaPHRA7B/alignment-newsletter-39",0,"",""],["Approval-directed agency and the decision theory of Newcomb-like problems","Caspar Oesterheld","2019","report","link.springer.com","link.springer.com/article/10.1007/s11229-019-02148-2",0,"","theory"],["Artificial Intelligence: American Attitudes and Trends","Baobao Zhang and Allan Dafoe","2019","report","ssrn.com","www.ssrn.com/abstract=3312874",0,"",""],["Bridging near- and long-term concerns about AI","Stephen Cave and Seán S. Ó hÉigeartaigh","2019","report","nature.com","www.nature.com/articles/s42256-018-0003-2",0,"",""],["Digital Authoritarianism: Evolving Chinese And Russian Models","Shazeda Ahmed and 3 others","2019","report","jstor.org","www.jstor.org/stable/pdf/resrep19585.11.pdf",0,"",""],["Failure Modes in Machine Learning - Security documentation","Ram Shankar Siva Kumar and 4 others","2019","report","docs.microsoft.com","docs.microsoft.com/en-us/security/failure-modes-in-machine-learning",0,"",""],["Feasibility of Training an AGI using Deep RL: A Very Rough Estimate","Baeo Maltinsky and 2 others","2019","report","mediangroup.org","mediangroup.org/docs/Feasibility%20of%20Training%20an%20AGI%20using%20Deep%20Reinforcement%20Learning,%20A%20Very%20Rough%20Estimate.pdf",0,"",""],["How Useful Is Quantilization For Mitigating Specification-Gaming?","Ryan Carey","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/SafeML2019_paper_40.pdf",0,"",""],["Learning Reward Machines for Partially Observable Reinforcement Learning","Rodrigo Toro Icarte and 5 others","2019","report","cs.toronto.edu","www.cs.toronto.edu/~rntoro/docs/LRM_paper.pdf",0,"","agents policy"],["Lessons for Artificial Intelligence from Other Global Risks","Seth Baum","2019","report","gcrinstitute.org","gcrinstitute.org/papers/lessons.pdf",0,"",""],["Long-term trajectories of human civilization","Seth D. Baum and 9 others","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/trajectories.pdf",0,"",""],["Machine Learning Projects for Iterated Distillation and Ampliﬁcation","Owain Evans and 2 others","2019","report","owainevans.github.io","owainevans.github.io/pdfs/evans_ida_projects.pdf",0,"",""],["ObjectNet: A large-scale bias-controlled dataset for pushing the limits of object recognition models","Andrei Barbu and 7 others","2019","report","papers.nips.cc","papers.nips.cc/paper_files/paper/2019/file/97af07a14cacba681feacf3012730892-Paper.pdf",0,"",""],["ObjectNet: A large-scale bias-controlled dataset for pushing the limits of object recognition models","Andrei Barbu and 7 others","2019","report","papers.nips.cc","papers.nips.cc/paper/2019/hash/97af07a14cacba681feacf3012730892-Abstract.html",0,"",""],["Optimization Regularization through Time Penalty","Linda Linsefors","2019","blog","AI Alignment Forum","www.alignmentforum.org/posts/ehLX2RdbD5ZkeJyuJ/optimization-regularization-through-time-penalty",0,"",""],["Reframing Superintelligence","Eric Drexler","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/reframing/",0,"","agents"],["Reframing Superintelligence: Comprehensive AI Services as General Intelligence","K Eric Drexler","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Reframing_Superintelligence_FHI-TR-2019-1.1-1.pdf",0,"",""],["Revisiting the Insights model","Median Group","2019","report","mediangroup.org","mediangroup.org/insights2.html",0,"",""],["Robust program equilibrium","Caspar Oesterheld","2019","report","link.springer.com","link.springer.com/article/10.1007/s11238-018-9679-3",0,"",""],["Stable Agreements in Turbulent Times: A Legal Toolkit for Constrained Temporal Decision Transmission","Cullen O’Keefe and J D Candidate","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Stable-Agreements.pdf",0,"",""],["Standards for AI Governance: International Standards to Enable Global Coordination in AI Research & Development","Peter Cihon","2019","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Standards_-FHI-Technical-Report.pdf",0,"","governance"],["Surveying Safety-relevant AI Characteristics","Jose Hernandez-Orallo and 2 others","2019","report","ceur-ws.org","ceur-ws.org/Vol-2301/paper_22.pdf",0,"",""],["The Bitter Lesson","Rich Sutton","2019","report","incompleteideas.net","www.incompleteideas.net/IncIdeas/BitterLesson.html",0,"","deception robustness"],["The Evidentialist’s Wager","William MacAskill and 4 others","2019","report","globalprioritiesinstitute.org","globalprioritiesinstitute.org/wp-content/uploads/MacAskill-Vallinder-Shulman-Osterheld-Treutlein_evidentialists-wager.pdf",0,"",""],["The Role and Limits of Principles in AI Ethics: Towards a Focus on Tensions","Jess Whittlestone and 3 others","2019","report","dl.acm.org","dl.acm.org/doi/pdf/10.1145/3306618.3314289",0,"",""],["There is plenty of time at the bottom: the economics, risk and ethics of time compression","Anders Sandberg","2019","report","ora.ox.ac.uk","ora.ox.ac.uk/objects/uuid:51ee3d43-533c-4de6-904e-be12c27afdca/download_file?file_format=pdf&safe_filename=There%2Bis%2Bplenty%2Bof%2Btime%2Bat%2Bthe%2Bbottom%2B4.pdf&type_of_work=Journal+article",0,"",""],["Impossibility and Uncertainty Theorems in AI Value Alignment (or why your AGI should not have a utility function)","Peter Eckersley","2018","paper","arXiv preprint","arxiv.org/abs/1901.00064",0,"",""],["Why I expect successful (narrow) alignment","Tobias_Baumann","2018","blog","EA Forum","forum.effectivealtruism.org/posts/NsHSu2wLWpgiALbwm/why-i-expect-successful-narrow-alignment",0,"",""],["Penalizing Impact via Attainable Utility Preservation","TurnTrout","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/mDTded2Dn7BKRBEPX/penalizing-impact-via-attainable-utility-preservation",0,"",""],["Reconciling modern machine learning practice and the bias-variance trade-off","Mikhail Belkin and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.11118",0,"",""],["An AI Race for Strategic Advantage: Rhetoric and Risks","Stephen Cave and Seán S. ÓhÉigeartaigh","2018","report","dl.acm.org","dl.acm.org/doi/10.1145/3278721.3278780",0,"",""],["Learning Not to Learn: Training Deep Neural Networks with Biased Data","Byungju Kim and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.10352",0,"",""],["Alignment Newsletter #38","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Mt9ZwedTgfeac4pD9/alignment-newsletter-38",0,"",""],["Can rationality be measured?","Tshilidzi Marwala","2018","paper","arXiv preprint","arxiv.org/abs/1812.10144",0,"",""],["Reinterpreting \"AI and Compute\"","habryka","2018","blog","LessWrong","www.lesswrong.com/posts/EjssJnp9fNhvdDEdK/reinterpreting-ai-and-compute",0,"","forecasting"],["[Link] Center for the Governance of AI (GovAI) Annual Report 2018","MarkusAnderljung","2018","blog","EA Forum","forum.effectivealtruism.org/posts/kMcr72cQ78q8G2jNv/link-center-for-the-governance-of-ai-govai-annual-report",0,"","governance"],["Anthropic probabilities and cost functions","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/uzb3u3zMTkrSEhCaf/anthropic-probabilities-and-cost-functions",0,"",""],["Human-AI Learning Performance in Multi-Armed Bandits","Ravi Pandya and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.09376",0,"","agents"],["The case for taking AI seriously as a threat to humanity","Kelsey Piper","2018","report","vox.com","www.vox.com/future-perfect/2018/12/21/18126576/ai-artificial-intelligence-machine-learning-safety-alignment",0,"",""],["Anthropic paradoxes transposed into Anthropic Decision Theory","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/PgsxXNSDsyz4DFEuw/anthropic-paradoxes-transposed-into-anthropic-decision",0,"","theory"],["Reasons compute may not drive AI capabilities growth","Tristan H","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/hSw4MNTc3gAwZWdx9/reasons-compute-may-not-drive-ai-capabilities-growth",0,"","forecasting"],["2018 AI Alignment Literature Review and Charity Comparison","Larks","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/a72owS5hz3acBK5xc/2018-ai-alignment-literature-review-and-charity-comparison",0,"",""],["Reinterpreting “AI and Compute”","Justis Mills","2018","blog","aiimpacts.org","aiimpacts.org/reinterpreting-ai-and-compute/",0,"",""],["Alignment Newsletter #37","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/wrMqGrGWaFfcAyvBW/alignment-newsletter-37",0,"",""],["Announcing a new edition of “Rationality: From AI to Zombies”","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/12/15/announcing-new-raz/",0,"",""],["December 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/12/16/december-2018-newsletter/",0,"",""],["The E-Coli Test for AI Alignment","johnswentworth","2018","blog","LessWrong","www.lesswrong.com/posts/ZdCztwnxXu3aC4kxZ/the-e-coli-test-for-ai-alignment",0,"",""],["The limit of artificial intelligence: Can machines be rational?","Tshilidzi Marwala","2018","paper","arXiv preprint","arxiv.org/abs/1812.06510",0,"",""],["Two Neglected Problems in Human-AI Safety","Wei Dai","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/HTgakSs6JpnogD6c2/two-neglected-problems-in-human-ai-safety",0,"",""],["Scaling shared model governance via model splitting","Miljan Martic and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.05979",0,"","governance robustness training-data"],["Critique of Superintelligence Part 1","Fods12","2018","blog","EA Forum","forum.effectivealtruism.org/posts/A8ndMGC4FTQq46RRX/critique-of-superintelligence-part-1",0,"",""],["Critique of Superintelligence Part 2","Fods12","2018","blog","EA Forum","forum.effectivealtruism.org/posts/BGWmAqrk64q2w6JjM/critique-of-superintelligence-part-2",0,"",""],["Critique of Superintelligence Part 3","Fods12","2018","blog","EA Forum","forum.effectivealtruism.org/posts/iKWbkomL8WrA8Yy4X/critique-of-superintelligence-part-3",0,"",""],["Critique of Superintelligence Part 4","Fods12","2018","blog","EA Forum","forum.effectivealtruism.org/posts/LLdHNTEHMoPYqGtHY/critique-of-superintelligence-part-4",0,"",""],["Critique of Superintelligence Part 5","Fods12","2018","blog","EA Forum","forum.effectivealtruism.org/posts/WhDa26A3AKaStvuD9/critique-of-superintelligence-part-5",0,"",""],["IRLAS: Inverse Reinforcement Learning for Architecture Search","Minghao Guo and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.05285",0,"","agents"],["Three AI Safety Related Ideas","Wei Dai","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/vbtvgNXkufFRSrx4j/three-ai-safety-related-ideas",0,"","scalable-oversight"],["Alignment Newsletter #36","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/XFEJg9gxak5agyxJo/alignment-newsletter-36",0,"",""],["Linking Artificial Intelligence Principles","Yi Zeng and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.04814",0,"",""],["Multi-agent predictive minds and AI alignment","Jan_Kulveit","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/3fkBWpE4f9nYbdf7E/multi-agent-predictive-minds-and-ai-alignment",0,"","agents"],["Assuming we've solved X, could we do Y...","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/95i5B78uhqyB3d6Xc/assuming-we-ve-solved-x-could-we-do-y",0,"",""],["Quantum immortality: Is decline of measure compensated by merging timelines?","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/5TYnquzAQPENyZXDa/quantum-immortality-is-decline-of-measure-compensated-by",0,"","forecasting theory"],["Building Ethically Bounded AI","Francesca Rossi and Nicholas Mattei","2018","paper","arXiv preprint","arxiv.org/abs/1812.03980",0,"","agents"],["Feature Denoising for Improving Adversarial Robustness","Cihang Xie and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.03411",0,"","deception robustness"],["Photos from the first AI Safety Camp","Kristina Němcová","2018","blog","aisafety.camp","aisafety.camp/2018/12/08/aisc1_photos/",0,"",""],["Photos from the second AI Safety Camp","Kristina Němcová","2018","blog","aisafety.camp","aisafety.camp/2018/12/08/aisc2-photos/",0,"",""],["AISC2: Research Summaries","Johannes","2018","blog","aisafety.camp","aisafety.camp/2018/12/07/aisc2-research-summaries/",0,"",""],["Building Ethics into Artificial Intelligence","Han Yu and 5 others","2018","paper","H. Yu, Z. Shen, C. Miao, C. Leung, V. R. Lesser & Q. Yang,\n  \"Building Ethics into Artificial Intelligence,\" in Proceedings of the 27th\n  International Joint Conference on Artificial Intelligence (IJCAI'18), pp.\n  5527-5533, 2018","arxiv.org/abs/1812.02953",0,"","governance"],["Off-Policy Deep Reinforcement Learning without Exploration","Scott Fujimoto and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.02900",0,"","agents policy"],["Toward the Engineering of Virtuous Machines","Naveen Sundar Govindarajulu and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.03868",0,"","agents"],["Verification of deep probabilistic models","Krishnamurthy Dvijotham and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.02795",0,"",""],["Factored Cognition","stuhlmueller","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/DFkGStzvj3jgXibFG/factored-cognition",0,"","scalable-oversight"],["Truly Autonomous Machines Are Ethical","John Hooker","2018","paper","arXiv preprint","arxiv.org/abs/1812.02217",0,"","deception"],["Why we need a *theory* of human values","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/zvrZi95EHqJPxdgps/why-we-need-a-theory-of-human-values",0,"",""],["Alignment Newsletter #35","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Pe3aqWXJWLHoB6vc4/alignment-newsletter-35",0,"",""],["Learning from Extrapolated Corrections","Jason Y. Zhang and Anca D. Dragan","2018","paper","arXiv preprint","arxiv.org/abs/1812.01225",0,"",""],["Rigorous Agent Evaluation: An Adversarial Approach to Uncover Catastrophic Failures","Jonathan Uesato and 9 others","2018","paper","arXiv preprint","arxiv.org/abs/1812.01647",0,"","evals agents"],["Coherence arguments do not entail goal-directed behavior","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/NxF5G6CJiof6cemTw/coherence-arguments-do-not-entail-goal-directed-behavior",0,"","theory"],["Benign model-free RL","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/PRaxzmDJdvie46ahL/benign-model-free-rl-1",0,"","scalable-oversight"],["Deep Learning Application in Security and Privacy -- Theory and Practice: A Position Paper","Julia A. Meister and 2 others","2018","paper","In WISTP 2018: Information Security Theory and Practice (pp.\n  129-144). Springer, Cham (2019)","arxiv.org/abs/1812.00190",0,"","evals governance"],["Intuitions about goal-directed behavior","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/DfcywmqRSkBaCB6Ma/intuitions-about-goal-directed-behavior",0,"","agents"],["Iterated Distillation and Amplification","Ajeya Cotra","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/HqLxuZ4LhaFhmAHWk/iterated-distillation-and-amplification-1",0,"","scalable-oversight"],["2017 in review","Malo Bourgon","2018","blog","intelligence.org","intelligence.org/2018/11/28/2017-in-review/",0,"",""],["Formal Open Problem in Decision Theory","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703753a9/formal-open-problem-in-decision-theory",0,"","theory"],["Hyperreal Brouwer","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754e4/hyperreal-brouwer",0,"",""],["Reflective oracles as a solution to the converse Lawvere problem","SamEisenstat","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037550d/reflective-oracles-as-a-solution-to-the-converse-lawvere",0,"",""],["The Ubiquitous Converse Lawvere Problem","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703753b9/the-ubiquitous-converse-lawvere-problem",0,"","theory"],["Bounded Oracle Induction","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/MgLeAWSeLbzx8mkZ2/bounded-oracle-induction",0,"","theory"],["MIRI’s newest recruit: Edward Kmett!","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/11/28/miris-newest-recruit-edward-kmett/",0,"",""],["Corrigibility","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/fkLYhTQteAu5SinAc/corrigibility",0,"","scalable-oversight instrumental-convergence agents robustness"],["Exploring Restart Distributions","Arash Tavakoli and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.11298",0,"","evals agents policy"],["MIRI’s 2018 Fundraiser","Malo Bourgon","2018","blog","intelligence.org","intelligence.org/2018/11/26/miris-2018-fundraiser/",0,"",""],["November 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/11/26/november-2018-newsletter/",0,"",""],["Alignment Newsletter #34","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZFT78ezD2yxLjo6QM/alignment-newsletter-34",0,"",""],["GAN Dissection: Visualizing and Understanding Generative Adversarial Networks","David Bau and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.10597",0,"","interpretability"],["Please Stop Explaining Black Box Models for High-Stakes Decisions","Cynthia Rudin","2018","paper","Nature Machine Intelligence, Vol 1, May 2019, 206-215","arxiv.org/abs/1811.10154",0,"","interpretability"],["Approval-directed bootstrapping","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/6x7oExXi32ot6HjJv/approval-directed-bootstrapping",0,"","scalable-oversight evals agents"],["Humans Consulting HCH","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/NXqs4nYXaq8q6dTTx/humans-consulting-hch",0,"","scalable-oversight evals"],["Fixed Point Discussion","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/mvqmY9MQ3qf88xRuM/fixed-point-discussion",0,"",""],["Explicability? Legibility? Predictability? Transparency? Privacy? Security? The Emerging Landscape of Interpretable Agent Behavior","Tathagata Chakraborti and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.09722",0,"","interpretability agents robustness"],["Hierarchical visuomotor control of humanoids","Josh Merel and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.09656",0,"","agents"],["Representer Point Selection for Explaining Deep Neural Networks","Chih-Kuan Yeh and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.09720",0,"",""],["Robustness via curvature regularization, and vice versa","Seyed-Mohsen Moosavi-Dezfooli and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.09716",0,"","robustness"],["2018 Update: Our New Research Directions","Nate Soares","2018","blog","intelligence.org","intelligence.org/2018/11/22/2018-update-our-new-research-directions/",0,"",""],["Approval-directed agents","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/7Hr8t6xwuuxBTqADK/approval-directed-agents-1",0,"","scalable-oversight agents"],["Iteration Fixed Point Exercises","Scott Garrabrant and SamEisenstat","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/9a2asxypuNjCmga3p/iteration-fixed-point-exercises",0,"",""],["Oversight of Unsafe Systems via Dynamic Safety Envelopes","David Manheim","2018","paper","arXiv preprint","arxiv.org/abs/1811.09246",0,"","mechanistic-interpretability governance robustness"],["Some cruxes on impactful alternatives to AI policy work","richard_ngo","2018","blog","EA Forum","forum.effectivealtruism.org/posts/DW4FyzRTfBfNDWm6J/some-cruxes-on-impactful-alternatives-to-ai-policy-work",0,"","governance policy"],["Time for AI to cross the human performance range in diabetic retinopathy","Aysja Johnson","2018","blog","aiimpacts.org","aiimpacts.org/diabetic-retinopathy-as-a-case-study-in-time-for-ai-to-cross-the-range-of-human-performance/",0,"",""],["New safety research agenda: scalable agent alignment via reward modeling","Vika","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/HBGd34LKvXM9TxvNf/new-safety-research-agenda-scalable-agent-alignment-via",0,"","agents"],["Prosaic AI alignment","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/YTq4X6inEudiHkHDF/prosaic-ai-alignment",0,"","goodharts-law benchmarks forecasting"],["\"Taking AI Risk Seriously\" – Thoughts by Andrew Critch","Raemon","2018","blog","EA Forum","forum.effectivealtruism.org/posts/7DhEnxBqP62jHmsAx/taking-ai-risk-seriously-thoughts-by-andrew-critch",0,"",""],["Alignment Newsletter #33","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KJPLW9XTaF3WxKxmq/alignment-newsletter-33",0,"",""],["Deeper Interpretability of Deep Networks","Tian Xu and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.07807",0,"","interpretability deception"],["Guiding Policies with Language via Meta-Learning","John D. Co-Reyes and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.07882",0,"","agents policy"],["Reinforcement Learning and Inverse Reinforcement Learning with System 1 and System 2","Alexander Peysakhovich","2018","paper","arXiv preprint","arxiv.org/abs/1811.08549",0,"","agents"],["Safely Probabilistically Complete Real-Time Planning and Exploration in Unknown Environments","David Fridovich-Keil and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.07834",0,"",""],["Scalable agent alignment via reward modeling: a research direction","Jan Leike and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.07871",0,"","deception agents"],["Diagonalization Fixed Point Exercises","Scott Garrabrant and SamEisenstat","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/FZkLa3GRLW97fpknG/diagonalization-fixed-point-exercises",0,"",""],["An unaligned benchmark","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZHXutm7KpoWEj9G2s/an-unaligned-benchmark",0,"","scalable-oversight benchmarks policy robustness"],["Fixed Point Exercises","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/mojJ6Hpri8rfzY78b/fixed-point-exercises",0,"","agents theory"],["Topological Fixed Point Exercises","Scott Garrabrant and SamEisenstat","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/svE3S6NKdPYoGepzq/topological-fixed-point-exercises",0,"",""],["Is Clickbait Destroying Our General Intelligence?","Eliezer Yudkowsky","2018","blog","LessWrong","www.lesswrong.com/posts/YicoiQurNBxSp7a65/is-clickbait-destroying-our-general-intelligence",0,"","goodharts-law"],["Clarifying \"AI Alignment\"","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZeE7EKHTFMBs8eMxn/clarifying-ai-alignment",0,"","robustness"],["Economics of Human-AI Ecosystem: Value Bias and Lost Utility in Multi-Dimensional Gaps","Daniel Muller","2018","paper","arXiv preprint","arxiv.org/abs/1811.06606",0,"","agents"],["Embedded Agency (full-text version)","Scott Garrabrant and abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/i3BTagvt3HbPMx6PN/embedded-agency-full-text-version",0,"","goodharts-law agents theory"],["Guiding the One-to-one Mapping in CycleGAN via Optimal Transport","Guansong Lu and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.06284",0,"",""],["Reward learning from human preferences and demonstrations in Atari","Borja Ibarz and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.06521",0,"","reward-hacking policy robustness"],["Switching hosting providers today, there probably will be some hiccups","habryka","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/amGDvs2ztj35mBF4Y/switching-hosting-providers-today-there-probably-will-be",0,"",""],["Woulda, Coulda, Shoulda: Counterfactually-Guided Policy Search","Lars Buesing and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.06272",0,"","evals policy"],["Emergence of Addictive Behaviors in Reinforcement Learning Agents","Vahid Behzadan and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.05590",0,"","reward-hacking evals agents"],["Natural Environment Benchmarks for Reinforcement Learning","Amy Zhang and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.06032",0,"","evals benchmarks"],["Acknowledging Human Preference Types to Support Value Learning","Nandi Sabrina Erin","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/mSPsyEwaymS74unND/acknowledging-human-preference-types-to-support-value",0,"",""],["Kelly bettors","DanielFilan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/iWXQgwpksstozSDeA/kelly-bettors",0,"",""],["The Steering Problem","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/4iPBctHSeHx8AkS6Z/the-steering-problem",0,"",""],["Alignment Newsletter #32","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/uHE9b4HWQjuqnFYqP/alignment-newsletter-32",0,"",""],["Improving Generalization for Abstract Reasoning Tasks Using Disentangled Feature Representations","Xander Steenbrugge and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.04784",0,"",""],["Learning Latent Dynamics for Planning from Pixels","Danijar Hafner and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.04551",0,"","agents"],["Future directions for ambitious value learning","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/EhNCnCkmu7MwrQ7yz/future-directions-for-ambitious-value-learning",0,"",""],["Towards Governing Agent's Efficacy: Action-Conditional $β$-VAE for Deep Transparent Reinforcement Learning","John Yang and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.04350",0,"","interpretability agents governance policy"],["AGI-11 survey","Justis Mills","2018","blog","aiimpacts.org","aiimpacts.org/agi-11-survey/",0,"",""],["Formal Limitations on the Measurement of Mutual Information","David McAllester and Karl Stratos","2018","paper","arXiv preprint","arxiv.org/abs/1811.04251",0,"",""],["Preface to the sequence on iterated amplification","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/HCv2uwgDGf5dyX5y6/preface-to-the-sequence-on-iterated-amplification",0,"","scalable-oversight"],["Specification gaming examples in AI","Samuel Rødal","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/nvkiGW4vH8CCHfoNi/specification-gaming-examples-in-ai",0,"","specification-gaming goodharts-law"],["A generic framework for privacy preserving deep learning","Theo Ryffel and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.04017",0,"",""],["Current AI Safety Roles for Software Engineers","ozziegooen","2018","blog","LessWrong","www.lesswrong.com/posts/3u8oZEEayqqjjZ7Nw/current-ai-safety-roles-for-software-engineers",0,"",""],["Model Mis-specification and Inverse Reinforcement Learning","Owain_Evans and jsteinhardt","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/cnC2RMWEGiGpJv8go/model-mis-specification-and-inverse-reinforcement-learning",0,"","robustness"],["A Geometric Perspective on the Transferability of Adversarial Directions","Zachary Charles and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.03531",0,"",""],["Embedded Curiosities","Scott Garrabrant and abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/j9CbmSsnprxB2uFY9/embedded-curiosities",0,"","deception agents theory"],["Embedded Curiosities","Abram Demski","2018","blog","intelligence.org","intelligence.org/2018/11/08/embedded-curiosities/",0,"",""],["Intrinsic Geometric Vulnerability of High-Dimensional Artificial Intelligence","Luca Bortolussi and Guido Sanguinetti","2018","paper","arXiv preprint","arxiv.org/abs/1811.03571",0,"","robustness"],["Learning from Demonstration in the Wild","Feryal Behbahani and 10 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.03516",0,"",""],["Stovepiping and Malicious Software: A Critical Review of AGI Containment","Jason M. Pittman and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.03653",0,"",""],["Integrative Biological Simulation, Neuropsychology, and AI Safety","Gopal P. Sarma and 2 others","2018","paper","Proceedings of the AAAI Workshop on Artificial Intelligence Safety\n  2019 co-located with the Thirty-Third AAAI Conference on Artificial\n  Intelligence 2019 (AAAI 2019)","arxiv.org/abs/1811.03493",0,"",""],["Latent Variables and Model Mis-Specification","jsteinhardt","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/gnvrixhDfG7S2TpNL/latent-variables-and-model-mis-specification",0,"","robustness"],["A Closer Look at Deep Policy Gradients","Andrew Ilyas and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.02553",0,"","policy"],["A Model for General Intelligence","Paul Yaworsky","2018","paper","arXiv preprint","arxiv.org/abs/1811.02546",0,"","deception"],["An Optimal Itinerary Generation in a Configuration Space of Large Intellectual Agent Groups with Linear Logic","Dmitry Maximov","2018","paper","Advances in Systems Science and Applications. 2019. Vol. 19, No 4.\n  P. 79-86 https://ijassa.ipu.ru/index.php/ijassa/article/view/829/513","arxiv.org/abs/1811.02216",0,"","agents"],["MixTrain: Scalable Training of Verifiably Robust Neural Networks","Shiqi Wang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.02625",0,"","evals robustness training-data"],["Solomon's Code: Humanity in a World of Thinking Machines","Olaf Groth and Mark Nitzberg","2018","report","goodreads.com","www.goodreads.com/book/show/38471807-solomon-s-code",0,"",""],["Subsystem Alignment","abramdemski and Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ChierESmenTtCQqZy/subsystem-alignment",0,"","reward-hacking goodharts-law agents robustness theory"],["Subsystem Alignment","Scott Garrabrant","2018","blog","intelligence.org","intelligence.org/2018/11/06/embedded-subsystems/",0,"",""],["Alignment Newsletter #31","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/gvAFSv7Gtcwcbst32/alignment-newsletter-31",0,"",""],["Humans can be assigned any values whatsoever…","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ANupXf8XfZo2EJxGv/humans-can-be-assigned-any-values-whatsoever",0,"","policy"],["Beliefs at different timescales","Nisan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yf4KcTyk2hoXZh9x4/beliefs-at-different-timescales",0,"",""],["Explaining Explanations in AI","Brent Mittelstadt and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1811.01439",0,"","interpretability robustness"],["Robust Delegation","abramdemski and Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/iTpLAaPamcKyjmbFC/robust-delegation",0,"","scalable-oversight reward-hacking goodharts-law evals agents theory"],["Robust Delegation","Abram Demski","2018","blog","intelligence.org","intelligence.org/2018/11/04/embedded-delegation/",0,"",""],["When does rationality-as-search have nontrivial implications?","nostalgebraist","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/BGcXEijZ6HLnASNit/when-does-rationality-as-search-have-nontrivial-implications",0,"","theory"],["A Marauder's Map of Security and Privacy in Machine Learning","Nicolas Papernot","2018","paper","arXiv preprint","arxiv.org/abs/1811.01134",0,"","robustness"],["The easy goal inference problem is still hard","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/h9DesGT3WT9u2k7Hr/the-easy-goal-inference-problem-is-still-hard",0,"","policy"],["Embedded World-Models","abramdemski and Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/efWfvrWLgJmbBAs3m/embedded-world-models",0,"","agents theory"],["Embedded World-Models","Scott Garrabrant","2018","blog","intelligence.org","intelligence.org/2018/11/02/embedded-models/",0,"",""],["CHAI Newsletter 2018","CHAI","2018","report","drive.google.com","drive.google.com/file/d/11wx4NIdiM-ue9blBoJCJMqUqpehJDc_K/view?usp=sharing",0,"",""],["Discussion on the machine learning approach to AI safety","Vika","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5GFn87cmw7A5hzR89/discussion-on-the-machine-learning-approach-to-ai-safety",0,"","evals"],["Discussion on the machine learning approach to AI safety","Victoria Krakovna","2018","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2018/11/01/discussion-on-the-machine-learning-approach-to-ai-safety/",0,"",""],["Meta-execution","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/4GXqbNvpJ4hJtcoSX/meta-execution",0,"","scalable-oversight"],["What is ambitious value learning?","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5eX8ko7GCxwR5N9mN/what-is-ambitious-value-learning",0,"",""],["Decision Theory","abramdemski and Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/zcPLNNw4wgBX5k8kQ/decision-theory",0,"","evals agents robustness theory"],["Decision Theory","Abram Demski","2018","blog","intelligence.org","intelligence.org/2018/10/31/embedded-decisions/",0,"","theory"],["October 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/10/29/october-2018-newsletter/",0,"",""],["On the Effectiveness of Interval Bound Propagation for Training Verifiably Robust Models","Sven Gowal and 8 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.12715",0,"",""],["Preface to the sequence on value learning","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/oH8KMnXHnw964QyS6/preface-to-the-sequence-on-value-learning",0,"",""],["The Responsibility Quantification (ResQu) Model of Human Interaction with Automation","Nir Douer and Joachim Meyer","2018","paper","IEEE Transactions on Automation Science and Engineering, 17 (2),\n  1044-1060 (2020)","arxiv.org/abs/1810.12644",0,"","evals agents policy"],["Alignment Newsletter #30","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/SPa6YYyeam2exxPMy/alignment-newsletter-30",0,"",""],["Announcing the new AI Alignment Forum","Guest","2018","blog","intelligence.org","intelligence.org/2018/10/29/announcing-the-ai-alignment-forum/",0,"",""],["Assessing Generalization in Deep Reinforcement Learning","Charles Packer and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.12282",0,"","evals benchmarks agents"],["Embedded Agents","abramdemski and Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/p7x32SEt43ZMC9r7r/embedded-agents",0,"","agents theory"],["Embedded Agents","Scott Garrabrant","2018","blog","intelligence.org","intelligence.org/2018/10/29/embedded-agents/",0,"","agents"],["Introducing the AI Alignment Forum (FAQ)","habryka and 3 others","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/FoiiRDC3EhjHx7ayY/introducing-the-ai-alignment-forum-faq",0,"","scalable-oversight theory"],["Efficiently Combining Human Demonstrations and Interventions for Safe Training of Autonomous Systems in Real-Time","Vinicius G. Goecks and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.11545",0,"","agents"],["Neural Modular Control for Embodied Question Answering","Abhishek Das and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.11181",0,"","interpretability benchmarks policy"],["Mimetic vs Anchored Value Alignment in Artificial Intelligence","Tae Wan Kim and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.11116",0,"",""],["One-Shot Hierarchical Imitation Learning of Compound Visuomotor Tasks","Tianhe Yu and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.11043",0,"",""],["Inverse reinforcement learning for video games","Aaron Tucker and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.10593",0,"","deception training-data"],["Toward an AI Physicist for Unsupervised Learning","Tailin Wu and Max Tegmark","2018","paper","Phys. Rev. E 100, 033311 (2019)","arxiv.org/abs/1810.10525",0,"","agents"],["Thoughts on short timelines","Tobias_Baumann","2018","blog","EA Forum","forum.effectivealtruism.org/posts/b3goLZcNxt68WbHmB/thoughts-on-short-timelines",0,"","forecasting"],["Alignment Newsletter #29","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yKoW5bZjnJemEkPgc/alignment-newsletter-29",0,"",""],["Applying Deep Learning To Airbnb Search","Malay Haldar and 10 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.09591",0,"","deception"],["Do Deep Generative Models Know What They Don't Know?","Eric Nalisnick and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.09136",0,"","evals robustness training-data"],["The role of existing institutions in AI strategy _ Jade Leung _ Seth Baum-by Centre for Effective Altruism-video_id pgiwvmY3brg-date 20181023","Jade Leung and Seth Baum","2018","report","drive.google.com","drive.google.com/file/d/17BFY3y4hNBaE8ZqjIYne0-LzNvgsNs2s/view?usp=share_link",0,"",""],["Addressing three problems with counterfactual corrigibility: bad bets, defending against backstops, and overconfidence.","RyanCarey","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/owdBiF8pj6Lpwwdup/addressing-three-problems-with-counterfactual-corrigibility",0,"",""],["Safe Reinforcement Learning with Model Uncertainty Estimates","Björn Lütjens and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08700",0,"","agents policy"],["Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","Natasha Jaques and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08647",0,"","agents"],["BabyAI: A Platform to Study the Sample Efficiency of Grounded Language Learning","Maxime Chevalier-Boisvert and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08272",0,"",""],["Establishing Appropriate Trust via Critical States","Sandy H. Huang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08174",0,"","policy"],["Expressing Robot Incapability","Minae Kwon and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.08167",0,"","evals"],["O2A: One-shot Observational learning with Action vectors","Leo Pauly and 3 others","2018","paper","Front. Robot. AI 8:686368 (2021)","arxiv.org/abs/1810.07483",0,"",""],["Discriminator Rejection Sampling","Samaneh Azadi and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.06758",0,"","deception"],["Finding Options that Minimize Planning Time","Yuu Jinnai and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.07311",0,"","evals"],["On the Future: Prospects for Humanity","Martin Rees","2018","report","goodreads.com","www.goodreads.com/book/show/39204073-on-the-future",0,"",""],["Alignment Newsletter #28","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/3JfrwRNgSqH9fqsQT/alignment-newsletter-28",0,"",""],["CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","Cédric Colas and 4 others","2018","paper","Proceedings of the 36th International Conference on Machine\n  Learning 2019","arxiv.org/abs/1810.06284",0,"","agents policy robustness"],["Deep Imitative Models for Flexible Inference, Planning, and Control","Nicholas Rhinehart and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.06544",0,"","interpretability"],["Factorized Machine Self-Confidence for Decision-Making Agents","Brett W Israelsen and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.06519",0,"","agents assurance"],["Optimizing Agent Behavior over Long Time Scales by Transporting Value","Chia-Chun Hung and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.06721",0,"","evals agents"],["Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","David Janz and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.06530",0,"","benchmarks"],["Hierarchical Game-Theoretic Planning for Autonomous Vehicles","Jaime F. Fisac and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.05766",0,"","deception"],["BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","Jacob Devlin and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.04805",0,"",""],["Characterizing Adversarial Examples Based on Spatial Consistency Information for Semantic Segmentation","Chaowei Xiao and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.05162",0,"","robustness"],["Learning under Misspecified Objective Spaces","Andreea Bobu and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.05157",0,"","rlhf evals"],["Batch Active Preference-Based Learning of Reward Functions","Erdem Bıyık and Dorsa Sadigh","2018","paper","arXiv preprint","arxiv.org/abs/1810.04303",0,"",""],["Secure Deep Learning Engineering: A Software Quality Assurance Perspective","Lei Ma and 9 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.04538",0,"","interpretability assurance"],["Some cruxes on impactful alternatives to AI policy work","Richard_Ngo","2018","blog","LessWrong","www.lesswrong.com/posts/DJB82jKwgJE5NsWgT/some-cruxes-on-impactful-alternatives-to-ai-policy-work",0,"","governance policy"],["Standard ML Oracles vs Counterfactual ones","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/hJaJw6LK39zpyCKW6/standard-ml-oracles-vs-counterfactual-ones",0,"",""],["Alignment Newsletter #27","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/g7yj6afwKf5GurCyZ/alignment-newsletter-27",0,"",""],["Ethical Reflections on Artificial Intelligence","Brian Patrick Green","2018","report","apcz.umk.pl","apcz.umk.pl/czasopisma/index.php/SetF/article/view/SetF.2018.015",0,"",""],["Fast Context Adaptation via Meta-Learning","Luisa M Zintgraf and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.03642",0,"","interpretability benchmarks"],["Meta-Learning: A Survey","Joaquin Vanschoren","2018","paper","arXiv preprint","arxiv.org/abs/1810.03548",0,"",""],["Sanity Checks for Saliency Maps","Julius Adebayo and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.03292",0,"","evals training-data"],["The 30-Year Cycle In The AI Debate","Jean-Marie Chauvet","2018","paper","arXiv preprint","arxiv.org/abs/1810.04053",0,"","deception"],["PPO-CMA: Proximal Policy Optimization with Covariance Matrix Adaptation","Perttu Hämäläinen and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.02541",0,"","benchmarks policy robustness"],["A Rationality Condition for CDT Is That It Equal EDT (Part 1)","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/XW6Qi2LitMDb2MF8c/a-rationality-condition-for-cdt-is-that-it-equal-edt-part-1",0,"","theory"],["Episodic Curiosity through Reachability","Nikolay Savinov and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.02274",0,"","agents"],["The Rocket Alignment Problem","Eliezer Yudkowsky","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Gg9a4y8reWKtLe3Tn/the-rocket-alignment-problem",0,"","agents theory"],["Unsupervised Learning via Meta-Learning","Kyle Hsu and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.02334",0,"","robustness"],["The Rocket Alignment Problem","Eliezer Yudkowsky","2018","blog","intelligence.org","intelligence.org/2018/10/03/rocket-alignment/",0,"",""],["Alignment Newsletter #26","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ZANm3Sbu5RxRca2zt/alignment-newsletter-26",0,"",""],["Near-Optimal Representation Learning for Hierarchical Reinforcement Learning","Ofir Nachum and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.01257",0,"","policy"],["Paul Christiano on how OpenAI is developing real solutions to the 'AI alignment problem', and his vision of how humanity will progressively hand over decision-making to AI systems","80000_Hours","2018","blog","EA Forum","forum.effectivealtruism.org/posts/fmk8xJG2TPBc2W7zo/paul-christiano-on-how-openai-is-developing-real-solutions",0,"",""],["Reinforcement Learning with Perturbed Rewards","Jingkang Wang and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.01032",0,"","agents"],["Bayesian Policy Optimization for Model Uncertainty","Gilwoo Lee and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.01014",0,"","agents policy"],["Planning for cars that coordinate with people: leveraging effects on human actions for planning and active information gathering over human internal state","Dorsa Sadigh and 4 others","2018","report","link.springer.com","link.springer.com/10.1007/s10514-018-9746-1",0,"",""],["SmartChoices: Hybridizing Programming and Machine Learning","Victor Carbune and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.00619",0,"","deception"],["Variational Discriminator Bottleneck: Improving Imitation Learning, Inverse RL, and GANs by Constraining Information Flow","Xue Bin Peng and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.00821",0,"","evals"],["EDT solves 5 and 10 with conditional oracles","jessicata","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Rcwv6SPsmhkgzfkDw/edt-solves-5-and-10-with-conditional-oracles",0,"","theory"],["Few-Shot Goal Inference for Visuomotor Learning and Planning","Annie Xie and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.00482",0,"",""],["Leto among the Machines","Virgil Kurkjian","2018","blog","LessWrong","www.lesswrong.com/posts/TifG2m7BYW2sGmAoR/leto-among-the-machines",0,"","goodharts-law"],["September 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/09/30/september-2018-newsletter/",0,"",""],["Stakeholders in Explainable AI","Alun Preece and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1810.00184",0,"","interpretability"],["Training Machine Learning Models by Regularizing their Explanations","Andrew Slavin Ross","2018","paper","arXiv preprint","arxiv.org/abs/1810.00869",0,"","interpretability training-data"],["On the (in)applicability of corporate rights cases to digital minds","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/on-the-inapplicability-of-corporate-rights-cases-to-digital-minds/",0,"",""],["Asymptotic Decision Theory (Improved Writeup)","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXCvYqTZCsfN7WRrg/asymptotic-decision-theory-improved-writeup",0,"","agents theory"],["Building safe artificial intelligence: specification, robustness, and assurance","Pedro Ortega and Vishal Maini","2018","report","medium.com","medium.com/@deepmindsafetyresearch/building-safe-artificial-intelligence-52f5f75058f1",0,"","interpretability agents assurance robustness monitoring"],["Few-Shot Intent Inference via Meta-Inverse Reinforcement Learning","Kelvin Xu and 4 others","2018","report","proceedings.mlr.press","proceedings.mlr.press/v97/xu19d/xu19d.pdf",0,"",""],["Inferring Reward Functions from Demonstrators with Unknown Biases","Rohin Shah and 3 others","2018","report","openreview.net","openreview.net/forum?id=rkgqCiRqKQ",0,"",""],["New DeepMind AI Safety Research Blog","Vika","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Y8ezvHWZXW232CkKr/new-deepmind-ai-safety-research-blog",0,"",""],["Uncovering Surprising Behaviors in Reinforcement Learning via Worst-case Analysis","Avraham Ruderman and 7 others","2018","report","openreview.net","openreview.net/forum?id=SkgZNnR5tX",0,"",""],["Adding Neural Network Controllers to Behavior Trees without Destroying Performance Guarantees","Christopher Iliffe Sprague and Petter Ögren","2018","paper","arXiv preprint","arxiv.org/abs/1809.10283",0,"","agents"],["An unaligned benchmark","Paul Christiano","2018","report","ai-alignment.com","ai-alignment.com/an-unaligned-benchmark-b49ad992940b",0,"","scalable-oversight rlhf benchmarks deception robustness"],["Wireheading as a potential problem with the new impact measure","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/6EMdmeosYPdn74wuG/wireheading-as-a-potential-problem-with-the-new-impact",0,"","reward-hacking agents"],["Alignment Newsletter #25","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/s8JuDTo8mTcbHMcLW/alignment-newsletter-25",0,"",""],["Deep learning - deeper flaws?","Richard_Ngo","2018","blog","LessWrong","www.lesswrong.com/posts/mbQSrox38WyT8c6tQ/deep-learning-deeper-flaws",0,"","forecasting"],["Reflective AIXI and Anthropics","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/K8FTuEdAbsHDDw3hR/reflective-aixi-and-anthropics",0,"",""],["(Some?) Possible Multi-Agent Goodhart Interactions","Davidmanheim","2018","blog","LessWrong","www.lesswrong.com/posts/9evYBqHAvKGR3rxMC/some-possible-multi-agent-goodhart-interactions",0,"","goodharts-law agents"],["Interpretable Multi-Objective Reinforcement Learning through Policy Orchestration","Ritesh Noothigattu and 8 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.08343",0,"","interpretability agents governance policy"],["In Logical Time, All Games are Iterated Games","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/dKAJqBDZRMMsaaYo5/in-logical-time-all-games-are-iterated-games",0,"","agents theory"],["Playing the Game of Universal Adversarial Perturbations","Julien Perolat and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.07802",0,"","robustness"],["Quantum theory cannot consistently describe the use of itself","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/pxpiGtyZpxmXg8hHW/quantum-theory-cannot-consistently-describe-the-use-of",0,"","theory"],["Bridging syntax and semantics, empirically","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/EEPdbtvW8ei9Yi2e8/bridging-syntax-and-semantics-empirically",0,"",""],["Interpretable Reinforcement Learning with Ensemble Methods","Alexander Brown and Marek Petrik","2018","paper","arXiv preprint","arxiv.org/abs/1809.06995",0,"","interpretability"],["TStarBots: Defeating the Cheating Level Builtin AI in StarCraft II in the Full Game","Peng Sun and 10 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.07193",0,"","deception agents"],["Towards a New Impact Measure","TurnTrout","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yEa7kwoMpsBgaBCgb/towards-a-new-impact-measure",0,"","agents"],["Adversarial Imitation via Variational Inverse Reinforcement Learning","Ahmed H. Qureshi and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.06404",0,"","evals policy"],["Alignment Newsletter #24","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/YTG348f4pEYcicwQ3/alignment-newsletter-24",0,"",""],["Realism about rationality","Richard_Ngo","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/suxvE2ddnYMPJN9HD/realism-about-rationality",0,"",""],["Towards Better Interpretability in Deep Q-Networks","Raghuram Mandyam Annasamy and Katia Sycara","2018","paper","arXiv preprint","arxiv.org/abs/1809.05630",0,"","interpretability agents"],["Model-Based Reinforcement Learning via Meta-Policy Optimization","Ignasi Clavera and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.05214",0,"","policy"],["CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","Jiachen Yang and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.05188",0,"","agents policy"],["Introducing the Unrestricted Adversarial Examples Challenge","Tom B Brown and Catherine Olsson","2018","report","ai.googleblog.com","ai.googleblog.com/2018/09/introducing-unrestricted-adversarial.html",0,"","robustness"],["(A -> B) -> A","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/qhsELHzAHFebRJE59/a-greater-than-b-greater-than-a",0,"","agents theory"],["Abstraction Learning","Fei Deng and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.03956",0,"",""],["Petrov corrigibility","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/4g29JgtbJ283iJ3Bh/petrov-corrigibility",0,"",""],["Working together to face humanity’s greatest threats: Introduction to The Future of Research on Catastrophic and Existential Risk.","Adrian Currie and 3 others","2018","report","repository.cam.ac.uk","www.repository.cam.ac.uk/handle/1810/280193",0,"",""],["Alignment Newsletter #23","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KAmEZMYE3QKGmKBNd/alignment-newsletter-23",0,"",""],["Disagreement with Paul: alignment induction","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/h3Fp3Erddwnm5uthZ/disagreement-with-paul-alignment-induction",0,"","scalable-oversight agents"],["Expert-augmented actor-critic for ViZDoom and Montezumas Revenge","Michał Garmulewicz and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.03447",0,"","evals agents robustness"],["Comment on decision theory","Rob Bensinger","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/uKbxi2EJ3KBNRDGpL/comment-on-decision-theory",0,"","interpretability agents robustness theory"],["Discriminator-Actor-Critic: Addressing Sample Inefficiency and Reward Bias in Adversarial Imitation Learning","Ilya Kostrikov and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.02925",0,"","policy"],["Training for Faster Adversarial Robustness Verification via Inducing ReLU Stability","Kai Y. Xiao and 3 others","2018","paper","International Conference on Learning Representations (ICLR) 2019","arxiv.org/abs/1809.03008",0,"","robustness"],["Neural Guided Constraint Logic Programming for Program Synthesis","Lisa Zhang and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.02840",0,"","agents"],["Learning Invariances for Policy Generalization","Remi Tachet and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.02591",0,"","evals agents policy"],["Challenges of Context and Time in Reinforcement Learning: Introducing Space Fortress as a Benchmark","Akshat Agarwal and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.02206",0,"","benchmarks"],["AI Governance: A Research Agenda","habryka","2018","blog","LessWrong","www.lesswrong.com/posts/yTMWjeRyHFCPbGXsk/ai-governance-a-research-agenda",0,"","governance"],["Counterfactuals and reflective oracles","Nisan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/pgTioHEzaSddx5csN/counterfactuals-and-reflective-oracles",0,"",""],["Reinforcement Learning under Threats","Victor Gallego and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.01560",0,"","agents"],["A Roadmap for Robust End-to-End Alignment","Lê Nguyên Hoang","2018","paper","arXiv preprint","arxiv.org/abs/1809.01036",0,"",""],["Recurrent World Models Facilitate Policy Evolution","David Ha and Jürgen Schmidhuber","2018","paper","arXiv preprint","arxiv.org/abs/1809.01999",0,"","agents policy"],["Alignment Newsletter #22","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/GvRb4m6jAvsrtwJGH/alignment-newsletter-22",0,"",""],["'The Hyperbolic Time Chamber & Brain Emulation'","Gwern Branwen","2018","blog","gwern.net","www.gwern.net/Hyperbolic-Time-Chamber.page",0,"",""],["Impact Measure Desiderata","TurnTrout","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/c2oM7qytRByv6ZFtz/impact-measure-desiderata",0,"",""],["Summer MIRI Updates","Malo Bourgon","2018","blog","intelligence.org","intelligence.org/2018/09/01/summer-miri-updates/",0,"",""],["Cooperative Oracles","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/SgkaXQn3xqJkGQ2D8/cooperative-oracles",0,"",""],["Governing Boring Apocalypses: A new typology of existential vulnerabilities and exposures for existential risk research","Hin-Yan Liu and 2 others","2018","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0016328717301623",0,"",""],["When wishful thinking works","AlexMennen","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KbCHcb8yyjAMFAAPJ/when-wishful-thinking-works",0,"","policy"],["Bottle Caps Aren't Optimisers","DanielFilan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/26eupx3Byc8swRS7f/bottle-caps-aren-t-optimisers",0,"",""],["VOI is Only Nonnegative When Information is Uncorrelated With Future Action","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KER27SxZssfmsusxy/voi-is-only-nonnegative-when-information-is-uncorrelated",0,"","agents theory"],["Do what we mean vs. do what we say","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/8Q5h6hyBXTEgC6EZf/do-what-we-mean-vs-do-what-we-say",0,"",""],["History of the Development of Logical Induction","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/iBBK4j6RWC7znEiDv/history-of-the-development-of-logical-induction",0,"","theory"],["HLAI 2018 Field Report","Gordon Seidoh Worley","2018","blog","LessWrong","www.lesswrong.com/posts/axsizR4vEX8qtuLpR/hlai-2018-field-report",0,"","forecasting"],["\"Why Tool AIs Want to Be Agent AIs\"","Gwern Branwen","2018","blog","gwern.net","www.gwern.net/Tool-AI.page",0,"","agents"],["August 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/08/27/august-2018-newsletter/",0,"",""],["Corrigibility doesn't always have a good action to take","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/nbhTzEosM9sqEvr6P/corrigibility-doesn-t-always-have-a-good-action-to-take",0,"","robustness"],["Cycle-of-Learning for Autonomous Systems from Human Interaction","Nicholas R. Waytowich and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.09572",0,"","evals agents"],["Alignment Newsletter #21","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/2NDt9DSPRbDoZPcKT/alignment-newsletter-21",0,"",""],["Reference Post: Formal vs. Effective Pre-Commitment","Chris_Leong","2018","blog","LessWrong","www.lesswrong.com/posts/Q8tyoaMFmW8R9w9db/reference-post-formal-vs-effective-pre-commitment",0,"","theory"],["Using expected utility for Good(hart)","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/urZzJPwHtjewdKKHc/using-expected-utility-for-good-hart",0,"","goodharts-law"],["Why Self-Attention? A Targeted Evaluation of Neural Machine Translation Architectures","Gongbo Tang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.08946",0,"","evals"],["The Social Cost of Strategic Classification","Smitha Milli and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.08460",0,"","robustness"],["Book Review: AI Safety and Security","Michaël Trazzi","2018","blog","LessWrong","www.lesswrong.com/posts/a4tcqr7QBAgMHLbcz/book-review-ai-safety-and-security",0,"",""],["Alignment Newsletter #20","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/XXmeWeY8ZzaLgmvAK/alignment-newsletter-20",0,"",""],["Life-Long Disentangled Representation Learning with Cross-Domain Latent Homologies","Alessandro Achille and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.06508",0,"",""],["Reducing collective rationality to individual optimization in common-payoff games using MCMC","jessicata","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/JKSS8GEu7DGX4YuxN/reducing-collective-rationality-to-individual-optimization",0,"","agents policy"],["Active Inverse Reward Design.","Sören Mindermann and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.03060",0,"","robustness"],["Multi-task Maximum Entropy Inverse Reinforcement Learning.","Adam Gleave and Oliver Habryka","2018","paper","arXiv preprint","arxiv.org/abs/1805.08882",0,"","policy robustness"],["Where Do You Think You’re Going?: Inferring Beliefs about Dynamics from Behavior.","Siddharth Reddy and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08010",0,"","evals deception"],["A Broader View on Bias in Automated Decision-Making: Reflecting on Epistemology and Dynamics.","Roel Dobbe and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.00553",0,"","deception training-data"],["Alignment Newsletter #19","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/WCpv3KH7hvRqLTKqw/alignment-newsletter-19",0,"",""],["Analyzing Inverse Problems with Invertible Neural Networks","Lynton Ardizzone and 8 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.04730",0,"",""],["Confidence-aware motion prediction for real-time collision avoidance.","David Fridovich-Keil and 10 others","2018","report","journals.sagepub.com","journals.sagepub.com/doi/abs/10.1177/0278364919859436",0,"",""],["Cost Functions for Robot Motion Style.","Allan Zhou and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1809.00092",0,"",""],["Courteous Autonomous Cars.","Liting Sun and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.02633",0,"",""],["Discrete-Continuous Mixtures in Probabilistic Programming: Generalized Semantics and Inference Algorithms.","Yi Wu and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.02027",0,"","robustness"],["Distill Update 2018","Distill Editors","2018","report","Distill","distill.pub/2018/editorial-update",0,"",""],["Evaluating the Stability of Non-Adaptive Trading in Continuous Double Auctions.","Mason Wright and Michael P and Wellman","2018","report","dl.acm.org","dl.acm.org/doi/10.5555/3237383.3237475",0,"","evals"],["Expert, Crowdsourced, and Machine Assessment of Suicide Risk via Online Postings.","Han-Chin Shing and 5 others","2018","report","aclweb.org","www.aclweb.org/anthology/W18-0603/",0,"",""],["Expressing Robot Incapability.","Minae Kwon and 4 others","2018","report","dl.acm.org","dl.acm.org/doi/10.1145/3171221.3171276",0,"",""],["Learning from Physical Human Corrections, One Feature at a Time.","Andrea Bajcsy and 6 others","2018","report","dl.acm.org","dl.acm.org/doi/10.1145/3171221.3171267",0,"",""],["Learning from Richer Human Guidance: Augmenting Comparison-Based Learning with Feature Queries.","Chandrayee Basu and 3 others","2018","report","dl.acm.org","dl.acm.org/doi/10.1145/3171221.3171284",0,"",""],["Learning Human Ergonomic Preferences for Handovers.","Aaron Bestick and 4 others","2018","report","par.nsf.gov","par.nsf.gov/servlets/purl/10063845",0,"",""],["Logical Counterfactuals & the Cooperation Game","Chris_Leong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/NcA3dMJoWWEN4BQet/logical-counterfactuals-and-the-cooperation-game",0,"",""],["Many-Goals Reinforcement Learning.","Vivek Veeriah and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.09605",0,"","agents policy"],["Minimax-regret querying on side effects for safe optimality in factored Markov decision processes.","Shun Zhang and 4 others","2018","report","dl.acm.org","dl.acm.org/doi/10.5555/3304652.3304685",0,"",""],["On Learning Intrinsic Rewards for Policy Gradient Methods.","Zeyu Zheng and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.06459",0,"","agents policy robustness"],["Overrepresentation of extreme events in decision making reflects rational use of cognitive resources.","Falk Lieder and 3 others","2018","report","psycnet.apa.org","psycnet.apa.org/record/2017-46556-001",0,"",""],["Planning for cars that coordinate with people: leveraging effects on human actions for planning and active information gathering over human internal state.","Dorsa Sadigh and 7 others","2018","report","link.springer.com","link.springer.com/article/10.1007/s10514-018-9746-1",0,"",""],["Rational metareasoning and the plasticity of cognitive control.","Falk Lieder and 4 others","2018","report","journals.plos.org","journals.plos.org/ploscompbiol/article?id=10.1371/journal.pcbi.1006043",0,"",""],["Sensitivity to Shared Information in Social Learning.","Andrew Whalen and 3 others","2018","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/full/10.1111/cogs.12485",0,"",""],["Social Cohesion in Autonomous Driving.","Nicholas C and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.03845",0,"",""],["SoK: Security and Privacy in Machine Learning.","Nicolas Papernot and 4 others","2018","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/8406613",0,"",""],["Solomon’s Code: Humanity in a World with Thinking Machines.","Olaf Graf and Mark Nitzberg","2018","report","cambrian.ai","www.cambrian.ai/our-book",0,"",""],["Using Trusted Data to Train Deep Networks on Labels Corrupted by Severe Noise.","Dan Hendrycks and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.05300",0,"","robustness training-data"],["Directed Policy Gradient for Safe Reinforcement Learning with Human Advice","Hélène Plisnier and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.04096",0,"","rlhf agents policy"],["Risk-Sensitive Generative Adversarial Imitation Learning","Jonathan Lacotte and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.04468",0,"","evals agents"],["Building Safer AGI by introducing Artificial Stupidity","Michaël Trazzi and Roman V. Yampolskiy","2018","paper","arXiv preprint","arxiv.org/abs/1808.03644",0,"",""],["A Note on the Existence of Ratifiable Acts.","Joseph Y and Halpern","2018","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/papers/newcomb.pdf",0,"",""],["A Regression Approach for Modeling Games with Many Symmetric Players.","Bryce Wiedenbeck and 3 others","2018","report","cs.swarthmore.edu","www.cs.swarthmore.edu/~bryce/publications/A_Regression_Approach_for_Modeling_Games_with_Many_Symmetric_Players.pdf",0,"",""],["Combining the Causal Judgments of Experts with Possibly Different Focus Areas.","Meir Friedenberg and 2 others","2018","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/papers/focus.pdf",0,"",""],["Empirical evidence for resource-rational anchoring and adjustment.","Falk Lieder and 7 others","2018","report","link.springer.com","link.springer.com/article/10.3758/s13423-017-1288-6",0,"",""],["Estimation with Incomplete Data: The Linear Case.","Karthika Mohan and 2 others","2018","report","ijcai.org","www.ijcai.org/Proceedings/2018/705",0,"",""],["Incentive-Compatible Mechanisms for Norm Monitoring in Open Multi-agent perspectives and applications.","Natasha Alechina and 5 others","2018","report","jair.org","www.jair.org/index.php/jair/article/view/11214",0,"","agents monitoring"],["Is state-dependent valuation more adaptive than simpler rules?.","Joseph Y and 2 others","2018","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0376635717302048",0,"",""],["Negotiable Reinforcement Learning for Pareto Optimal Sequential Decision-Making.","Nishant Desai and 3 others","2018","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/nips18-pareto.pdf",0,"",""],["On Handling Self-masking and Other Hard Missing Data Problems.","Karthika Mohan","2018","report","why19.causalai.net","why19.causalai.net/papers/mohan-why19.pdf",0,"",""],["Open Category Detection with PAC Guarantees.","Si Liu and 4 others","2018","report","proceedings.mlr.press","proceedings.mlr.press/v80/liu18e/liu18e.pdf",0,"",""],["The anchoring bias reflects rational use of cognitive resources.","Falk Lieder and 7 others","2018","report","link.springer.com","link.springer.com/article/10.3758/s13423-017-1286-8",0,"",""],["Adversarial Vision Challenge","Wieland Brendel and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.01976",0,"",""],["Alignment Newsletter #18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/9a33WfdPe9Cd26vL9/alignment-newsletter-18",0,"",""],["Moral realism and AI alignment","Caspar Oesterheld","2018","report","casparoesterheld.com","casparoesterheld.com/2018/08/06/moral-realism-and-ai-alignment/",0,"",""],["Is Robustness the Cost of Accuracy? -- A Comprehensive Study on the Robustness of 18 Deep Image Classification Models","Dong Su and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.01688",0,"","robustness"],["Generalization Error in Deep Learning","Daniel Jakubovitz and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.01174",0,"",""],["Learning Actionable Representations from Visual Observations","Debidatta Dwibedi and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.00928",0,"","agents policy"],["Neural Arithmetic Logic Units","Andrew Trask and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.00508",0,"",""],["Sandboxing by Physical Simulation?","moridinamael","2018","blog","LessWrong","www.lesswrong.com/posts/aXSrXgNS5D3Zstqtw/sandboxing-by-physical-simulation",0,"",""],["Counterfactuals, thick and thin","Nisan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/YSH3RFSFESzsa5Nrg/counterfactuals-thick-and-thin",0,"",""],["Safely and usefully spectating on AIs optimizing over toy worlds","AlexMennen","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ikN9qQEkrFuPtYd6Y/safely-and-usefully-spectating-on-ais-optimizing-over-toy",0,"",""],["Security and Privacy Issues in Deep Learning","Ho Bae and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.11655",0,"","deception robustness"],["Techniques for Interpretable Machine Learning","Mengnan Du and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1808.00033",0,"","interpretability evals"],["Alignment Newsletter #17","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/xKbtjfQ2y4anW3fZQ/alignment-newsletter-17",0,"",""],["Reinforced Auto-Zoom Net: Towards Accurate and Fast Breast Cancer Segmentation in Whole-slide Images","Nanqing Dong and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.11113",0,"","evals policy"],["A Gym Gridworld Environment for the Treacherous Turn","Michaël Trazzi","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/cKfryXvyJ522iFuNF/a-gym-gridworld-environment-for-the-treacherous-turn",0,"","instrumental-convergence"],["Decisions are not about changing the world, they are about learning what world you live in","shminux","2018","blog","LessWrong","www.lesswrong.com/posts/TQvSZ4n4BuntC22Af/decisions-are-not-about-changing-the-world-they-are-about",0,"","agents theory"],["TensorFuzz: Debugging Neural Networks with Coverage-Guided Fuzzing","Augustus Odena and Ian Goodfellow","2018","paper","arXiv preprint","arxiv.org/abs/1807.10875",0,"",""],["MDL Intelligence Distillation : Exploring Strategies for Safe Access to Superintelligent Problem-Solving Capabilities","K. Eric Drexler","2018","report","taylorfrancis.com","www.taylorfrancis.com/chapters/edit/10.1201/9781351251389-6/mdl-intelligence-distillation-eric-drexler",0,"",""],["Evaluating and Understanding the Robustness of Adversarial Logit Pairing","Logan Engstrom and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.10272",0,"","evals deception robustness"],["Multi-Agent Generative Adversarial Imitation Learning","Jiaming Song and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.09936",0,"","agents policy robustness"],["Variational Option Discovery Algorithms","Joshua Achiam and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.10299",0,"","agents policy"],["Differentiable Image Parameterizations","Alexander Mordvintsev and 3 others","2018","report","Distill","distill.pub/2018/differentiable-parameterizations",0,"",""],["July 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/07/25/july-2018-newsletter/",0,"",""],["Narrow AI Nanny: Reaching Strategic Advantage via Narrow AI to Prevent Creation of the Dangerous Superintelligence","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/7ysKDyQDPK3dDAbkT/narrow-ai-nanny-reaching-strategic-advantage-via-narrow-ai",0,"",""],["The Evil Genie Puzzle","Chris_Leong","2018","blog","LessWrong","www.lesswrong.com/posts/YSEtEtqf8hRBhKBS9/the-evil-genie-puzzle",0,"","theory"],["ISO: Name of Problem","johnswentworth","2018","blog","LessWrong","www.lesswrong.com/posts/NA5LYobKuFMKCvzMo/iso-name-of-problem",0,"",""],["Learning Plannable Representations with Causal InfoGAN","Thanard Kurutach and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.09341",0,"",""],["Alignment Newsletter #16: 07/23/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/63nvBi4ooCAsKCphw/alignment-newsletter-16-07-23-18",0,"",""],["Contrastive Explanations for Reinforcement Learning in terms of Expected Consequences","Jasper van der Waa and 3 others","2018","paper","IJCAI-18 Workshop on Explainable AI (XAI). Vol. 37. 2018","arxiv.org/abs/1807.08706",0,"","interpretability agents policy"],["Let's Discuss Functional Decision Theory","Chris_Leong","2018","blog","LessWrong","www.lesswrong.com/posts/2THFt7BChfCgwYDeA/let-s-discuss-functional-decision-theory",0,"","theory"],["EnsembleDAgger: A Bayesian Approach to Safe Imitation Learning","Kunal Menda and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.08364",0,"","policy training-data"],["Safe Option-Critic: Learning Safety in the Option-Critic Architecture","Arushi Jain and 2 others","2018","paper","The Knowledge Engineering Review 36 (2021) e4","arxiv.org/abs/1807.08060",0,"","agents policy"],["Stable Pointers to Value III: Recursive Quantilization","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/bEa4FuLS4r7hExoty/stable-pointers-to-value-iii-recursive-quantilization",0,"",""],["Can few-shot learning teach AI right from wrong?","Charlie Steiner","2018","blog","LessWrong","www.lesswrong.com/posts/5iAMEmDkvKqHH5LZc/can-few-shot-learning-teach-ai-right-from-wrong",0,"",""],["Knowledge Integration for Disease Characterization: A Breast Cancer Example","Oshani Seneviratne and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.07991",0,"","monitoring"],["Learning Heuristics for Quantified Boolean Formulas through Deep Reinforcement Learning","Gil Lederman and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.08058",0,"",""],["Probability is Real, and Value is Complex","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/oheKfWA7SsvpK7SGp/probability-is-real-and-value-is-complex",0,"",""],["Generalized Kelly betting","Linda Linsefors","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/97tCruegbz8GXRFzi/generalized-kelly-betting",0,"",""],["Backplay: \"Man muss immer umkehren\"","Cinjon Resnick and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.06919",0,"","agents policy robustness"],["Generative Adversarial Imitation from Observation","Faraz Torabi and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.06158",0,"","agents"],["Interpretable Latent Spaces for Learning from Demonstration","Yordan Hristov and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.06583",0,"","interpretability agents"],["Alignment Newsletter #15: 07/16/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/EQ9dBequfxmeYzhz6/alignment-newsletter-15-07-16-18",0,"",""],["Buridan's ass in coordination games","jessicata","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/4xpDnGaKz472qB4LY/buridan-s-ass-in-coordination-games",0,"",""],["Compact vs. Wide Models","Vaniver","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/JkCPkMxuftohieb8B/compact-vs-wide-models",0,"",""],["Hardware overhang","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/hardware-overhang/",0,"",""],["Introducing Quantum-Like Influence Diagrams for Violations of the Sure Thing Principle","Catarina Moreira and Andreas Wichert","2018","paper","Quantum Interactions, 2018","arxiv.org/abs/1807.06142",0,"",""],["Meta-Learning with Latent Embedding Optimization","Andrei A. Rusu and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.05960",0,"","evals deception"],["Safe Reinforcement Learning via Probabilistic Shields","Nils Jansen and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.06096",0,"",""],["Announcement: AI alignment prize round 3 winners and next round","cousin_it","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/juBRTuE3TLti5yB35/announcement-ai-alignment-prize-round-3-winners-and-next",0,"",""],["Deep Learning in the Wild","Thilo Stadelmann and 10 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.04950",0,"","monitoring"],["Exploring Hierarchy-Aware Inverse Reinforcement Learning","Chris Cundy and Daniel Filan","2018","paper","1st Workshop on Goal Specifications for Reinforcement Learning,\n  ICML 2018, Stockholm, Sweden, 2018","arxiv.org/abs/1807.05037",0,"","agents"],["Minimax-regret querying on side effects for safe optimality in factored Markov decision processes","Shun Zhang and 2 others","2018","report","web.eecs.umich.edu","web.eecs.umich.edu/~baveja/Papers/ijcai-2018.pdf",0,"","agents policy"],["Model Reconstruction from Model Explanations","Smitha Milli and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.05185",0,"","deception"],["An Agent is a Worldline in Tegmark V","komponisto","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/brQYmeX4HFrPbs4XP/an-agent-is-a-worldline-in-tegmark-v",0,"","agents"],["Historic trends in structure heights","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/discontinuity-from-the-burj-khalifa/",0,"",""],["The Bottleneck Simulator: A Model-based Deep Reinforcement Learning Approach","Iulian Vlad Serban and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.04723",0,"","evals deception policy"],["Visual Reinforcement Learning with Imagined Goals","Ashvin Nair and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.04742",0,"","agents policy"],["A comment on the IDA-AlphaGoZero metaphor; capabilities versus alignment","AlexMennen","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yXFKh2jGysQNfX2NM/a-comment-on-the-ida-alphagozero-metaphor-capabilities",0,"","scalable-oversight"],["Agents That Learn From Human Behavior Can't Learn Human Values That Humans Haven't Learned Yet","steven0461","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/DfewqowdzDdCD7S9y/agents-that-learn-from-human-behavior-can-t-learn-human",0,"","agents"],["An environment for studying counterfactuals","Nisan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/hhNH3knNHgdkonAKB/an-environment-for-studying-counterfactuals",0,"",""],["Are pre-specified utility functions about the real world possible in principle?","mlogan","2018","blog","LessWrong","www.lesswrong.com/posts/ZFuwgLfRH4qyFTMib/are-pre-specified-utility-functions-about-the-real-world",0,"","theory"],["Bounding Goodhart's Law","eric_langlois","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/nu2KcjGf2EYyY2oRJ/bounding-goodhart-s-law",0,"","goodharts-law"],["Clarifying Consequentialists in the Solomonoff Prior","vlad_m","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/jP3vRbtvDtBtgvkeb/clarifying-consequentialists-in-the-solomonoff-prior",0,"",""],["Complete Class: Consequentialist Foundations","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/sZuw6SGfmZHvcAAEP/complete-class-consequentialist-foundations",0,"","theory"],["Conceptual problems with utility functions","Dacyn","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Nx4DsTpMaoTiTp4RQ/conceptual-problems-with-utility-functions",0,"",""],["Conditions under which misaligned subagents can (not) arise in classifiers","anon1","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/xmzNAoWcYQfMv3j6J/conditions-under-which-misaligned-subagents-can-not-arise-in",0,"","agents"],["Decision-theoretic problems and Theories; An (Incomplete) comparative list","somervta","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/cWEhuXQBxRwxmhER5/decision-theoretic-problems-and-theories-an-incomplete",0,"","theory"],["Mathematical Mindset","komponisto","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/wxBBRzR4FS7nGBjbD/mathematical-mindset",0,"",""],["Mechanistic Transparency for Machine Learning","DanielFilan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/3kwR2dufdJyJamHQq/mechanistic-transparency-for-machine-learning",0,"","interpretability"],["No, I won't go there, it feels like you're trying to Pascal-mug me","Rupert","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/o7MXZgx3SGpqSxHYZ/no-i-won-t-go-there-it-feels-like-you-re-trying-to-pascal",0,"",""],["On the Role of Counterfactuals in Learning","Max Kanwal","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/MeYeLEr4RNGreJZcB/on-the-role-of-counterfactuals-in-learning",0,"",""],["A Game-Based Approximate Verification of Deep Neural Networks with Provable Guarantees","Min Wu and 4 others","2018","paper","Theoretical Computer Science 807 (2020) 298-329","arxiv.org/abs/1807.03571",0,"","evals robustness"],["A Simple Unified Framework for Detecting Out-of-Distribution Samples and Adversarial Attacks","Kimin Lee","2018","paper","arXiv preprint","arxiv.org/abs/1807.03888",0,"","evals deception robustness training-data"],["Announcing AlignmentForum.org Beta","Raemon","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/JiMAMNAb55Qq24nES/announcing-alignmentforum-org-beta",0,"",""],["Bayesian Probability is for things that are Space-like Separated from You","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/FvcyMMaJKhYibtFDD/bayesian-probability-is-for-things-that-are-space-like",0,"","theory"],["Conditioning, Counterfactuals, Exploration, and Gears","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/uQHAJ7TdBbweRR5iS/conditioning-counterfactuals-exploration-and-gears",0,"",""],["Interpreting AI compute trends","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/interpreting-ai-compute-trends/",0,"",""],["Logical Uncertainty and Functional Decision Theory","swordsintoploughshares","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/jyeYdBXAwsc4LPs7m/logical-uncertainty-and-functional-decision-theory",0,"","theory"],["Probability is fake, frequency is real","Linda Linsefors","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/vmfW2qTac4vF3YS3J/probability-is-fake-frequency-is-real",0,"",""],["Repeated (and improved) Sleeping Beauty problem","Linda Linsefors","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KZh9eKCkBRbevkGnP/repeated-and-improved-sleeping-beauty-problem",0,"",""],["Representation Learning with Contrastive Predictive Coding","Aaron van den Oord and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.03748",0,"","evals"],["Universal Transformers","Mostafa Dehghani and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.03819",0,"",""],["Alignment Newsletter #14","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yamyc3oXEpFb8kty6/alignment-newsletter-14",0,"",""],["Feature-wise transformations","Vincent Dumoulin and 6 others","2018","report","Distill","distill.pub/2018/feature-wise-transformations",0,"",""],["Troubling Trends in Machine Learning Scholarship","Zachary C. Lipton and Jacob Steinhardt","2018","paper","arXiv preprint","arxiv.org/abs/1807.03341",0,"",""],["Occasional update July 5 2018","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/occasional-update-july-5-2018/",0,"",""],["Benchmarking Neural Network Robustness to Common Corruptions and Surface Variations","Dan Hendrycks and Thomas G. Dietterich","2018","paper","arXiv preprint","arxiv.org/abs/1807.01697",0,"","evals benchmarks robustness"],["Ranked Reward: Enabling Self-Play Reinforcement Learning for Combinatorial Optimization","Alexandre Laterre and 8 others","2018","paper","Presented at the Thirty-second Conference on Neural Information\n  Processing Systems (NeurIPS 2018), Deep Reinforcement Learning Workshop,\n  Montreal, Canada, December 3-8, 2018","arxiv.org/abs/1807.01672",0,"","deception agents training-data"],["The Learning-Theoretic AI Alignment Research Agenda","Vanessa Kosoy","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375575/the-learning-theoretic-ai-alignment-research-agenda",0,"",""],["The Prediction Problem: A Variant on Newcomb's","Chris_Leong","2018","blog","LessWrong","www.lesswrong.com/posts/YpdTSt4kRnuSkn63c/the-prediction-problem-a-variant-on-newcomb-s",0,"","theory"],["Intertheoretic utility comparison","Stuart_Armstrong","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/hBJCMWELaW6MxinYW/intertheoretic-utility-comparison",0,"",""],["Alignment Newsletter #13: 07/02/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/rhdzxSfLXpxZRsHg6/alignment-newsletter-13-07-02-18",0,"",""],["Accurate Uncertainties for Deep Learning Using Calibrated Regression","","2018","paper","arXiv preprint","arxiv.org/abs/1807.00263",0,"","evals forecasting"],["Another take on agent foundations: formalizing zero-shot reasoning","zhukeepa","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/pu3ddLSZjjmiiqQfh/another-take-on-agent-foundations-formalizing-zero-shot",0,"","agents theory"],["Beyond Winning and Losing: Modeling Human Motivations and Behaviors Using Inverse Reinforcement Learning","Baoxiang Wang and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.00366",0,"","deception"],["Goodhart Taxonomy: Agreement","Ben Pace","2018","blog","LessWrong","www.lesswrong.com/posts/ZSkForf7e5nEKGDdb/goodhart-taxonomy-agreement-1",0,"","goodharts-law"],["Machine learning 2.0 : Engineering Data Driven AI Products","James Max Kanter and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.00401",0,"",""],["Minimax-Regret Querying on Side Effects for Safe Optimality in Factored Markov Decision Processes","Shun Zhang and 2 others","2018","report","ijcai.org","www.ijcai.org/proceedings/2018/676",0,"",""],["Paul's research agenda FAQ","zhukeepa","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Djs38EWYZG8o7JMWY/paul-s-research-agenda-faq",0,"","scalable-oversight"],["The Facets of Artificial Intelligence: A Framework to Track the Evolution of AI","Fernando Martínez-Plumed and 5 others","2018","report","ijcai.org","www.ijcai.org/proceedings/2018/718",0,"",""],["Towards Mixed Optimization for Reinforcement Learning with Program Synthesis","Surya Bhupatiraju and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1807.00403",0,"","policy"],["Modeling Friends and Foes","Pedro A. Ortega and Shane Legg","2018","paper","arXiv preprint","arxiv.org/abs/1807.00196",0,"","agents robustness"],["Overcoming Clinginess in Impact Measures","TurnTrout","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/DvmhXysefEyEvXuXS/overcoming-clinginess-in-impact-measures",0,"",""],["Policy Alignment","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/TeYro2ntqHNyQFx8r/policy-alignment",0,"","policy theory"],["“Cheating Death in Damascus” Solution to the Fermi Paradox","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/R9javXN9BN5nXWHZx/cheating-death-in-damascus-solution-to-the-fermi-paradox",0,"","theory"],["A Benchmark for Interpretability Methods in Deep Neural Networks","Sara Hooker and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.10758",0,"","interpretability benchmarks"],["Adversarial Reprogramming of Neural Networks","Gamaleldin F. Elsayed and 2 others","2018","paper","International Conference on Learning Representations 2019","arxiv.org/abs/1806.11146",0,"",""],["Illuminating Generalization in Deep Reinforcement Learning through Procedural Level Generation","Niels Justesen and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.10729",0,"","agents"],["New paper: “Forecasting using incomplete models”","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/06/27/forecasting-using-incomplete-models/",0,"","forecasting"],["Optimization Amplifies","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/zEvqFtT4AtTztfYC4/optimization-amplifies",0,"","goodharts-law"],["Adversarial Active Exploration for Inverse Dynamics Model Learning","Zhang-Wei Hong and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.10019",0,"","evals agents"],["Learning Existing Social Conventions via Observationally Augmented Self-Play","Adam Lerer and Alexander Peysakhovich","2018","paper","arXiv preprint","arxiv.org/abs/1806.10071",0,"","agents policy"],["Logical uncertainty and Mathematical uncertainty","AlexMennen","2018","blog","LessWrong","www.lesswrong.com/posts/q9xHFf8duqbc45YvT/logical-uncertainty-and-mathematical-uncertainty",0,"","theory"],["Multi-agent Inverse Reinforcement Learning for Certain General-sum Stochastic Games","Xiaomin Lin and 2 others","2018","paper","Journal of Artificial Intelligence Research 66 (2019), pp 473-502","arxiv.org/abs/1806.09795",0,"","benchmarks agents policy"],["The Alignment Newsletter #12: 06/25/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/MDSQEZeyakzAEKyGk/the-alignment-newsletter-12-06-25-18",0,"",""],["DARTS: Differentiable Architecture Search","Hanxiao Liu and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.09055",0,"","deception"],["UDT can learn anthropic probabilities","cousin_it","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/ma5Jc4wPT36j3X84P/udt-can-learn-anthropic-probabilities",0,"","theory"],["June 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/06/23/june-2018-newsletter/",0,"",""],["On Adversarial Examples for Character-Level Neural Machine Translation","Javid Ebrahimi and 2 others","2018","paper","COLING 2018","arxiv.org/abs/1806.09030",0,"","evals robustness"],["Human-Interactive Subgoal Supervision for Efficient Inverse Reinforcement Learning","Xinlei Pan and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.08479",0,"","evals deception agents"],["The Foundations of Deep Learning with a Path Towards General Intelligence","Eray Özkural","2018","paper","arXiv preprint","arxiv.org/abs/1806.08874",0,"",""],["Interpretable Discovery in Large Image Data Sets","Kiri L. Wagstaff and Jake Lee","2018","paper","arXiv preprint","arxiv.org/abs/1806.08340",0,"","interpretability"],["Interpretable to Whom? A Role-based Model for Analyzing Interpretable Machine Learning Systems","Richard Tomsett and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.07552",0,"","interpretability agents"],["RUDDER: Return Decomposition for Delayed Rewards","Jose A. Arjona-Medina and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.07857",0,"","policy"],["A Survey of Inverse Reinforcement Learning: Challenges, Methods and Progress","Saurabh Arora and Prashant Doshi","2018","paper","arXiv preprint","arxiv.org/abs/1806.06877",0,"","agents policy"],["The Alignment Newsletter #11: 06/18/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/aKhzD8m53oCiE38K7/the-alignment-newsletter-11-06-18-18",0,"",""],["Worrying about the Vase: Whitelisting","TurnTrout","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/H7KB44oKoSjSCkpzL/worrying-about-the-vase-whitelisting",0,"",""],["Evolving simple programs for playing Atari games","Dennis G Wilson and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.05695",0,"","evals benchmarks"],["Scrutinizing and De-Biasing Intuitive Physics with Neural Stethoscopes","Fabian B. Fuchs and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.05502",0,"",""],["Self-Imitation Learning","Junhyuk Oh and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.05635",0,"","agents policy robustness"],["Weak arguments against the universal prior being malign","X4vier","2018","blog","LessWrong","www.lesswrong.com/posts/Ecxevhvx85Y4eyFcu/weak-arguments-against-the-universal-prior-being-malign",0,"","agents"],["Counterfactual Mugging Poker Game","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/g3PwPgcdcWiP33pYn/counterfactual-mugging-poker-game",0,"","theory"],["The IQ of Artificial Intelligence","Dimiter Dobrev","2018","paper","Serdica Journal of Computing, Vol. 13, Number 1-2, 2019, pp.41-70","arxiv.org/abs/1806.04915",0,"",""],["Understanding the Meaning of Understanding","Daniele Funaro","2018","paper","arXiv preprint","arxiv.org/abs/1806.05234",0,"","robustness"],["Resource-Efficient Neural Architect","Yanqi Zhou and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.07912",0,"","policy"],["A general model of safety-oriented AI development","Wei Dai","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/idb5Ppp9zghcichJ5/a-general-model-of-safety-oriented-ai-development",0,"","scalable-oversight"],["Adaptive Mechanism Design: Learning to Promote Cooperation","Tobias Baumann and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.04067",0,"","agents"],["An Efficient, Generalized Bellman Update For Cooperative Inverse Reinforcement Learning","Dhruv Malik and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.03820",0,"","agents"],["Announcing the second AI Safety Camp","Lachouette","2018","blog","LessWrong","www.lesswrong.com/posts/WwsgJcey7nfXhZWfn/announcing-the-second-ai-safety-camp",0,"",""],["Defense Against the Dark Arts: An overview of adversarial example security research and future research directions","Ian Goodfellow","2018","paper","arXiv preprint","arxiv.org/abs/1806.04169",0,"","robustness"],["The Alignment Newsletter #10: 06/11/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Foqiq3TGmfYmQwrnH/the-alignment-newsletter-10-06-11-18",0,"",""],["Jade Leung and Seth Baum: The role of existing institutions in AI strategy","EA Global","2018","blog","EA Forum","forum.effectivealtruism.org/posts/Nrq9v3Kii7EmAhFk2/jade-leung-and-seth-baum-the-role-of-existing-institutions",0,"","governance policy"],["Quantum AI Box","Gurkenglas","2018","blog","LessWrong","www.lesswrong.com/posts/FWNpg7jYoECxsSegf/quantum-ai-box",0,"",""],["RFC: Meta-ethical uncertainty in AGI alignment","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/uR3znuBnaevssYDZY/rfc-meta-ethical-uncertainty-in-agi-alignment",0,"",""],["Beyond Astronomical Waste","Wei Dai","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Qz6w4GYZpgeDp6ATB/beyond-astronomical-waste",0,"","theory"],["Simplifying Reward Design through Divide-and-Conquer","Ellis Ratner and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.02501",0,"","deception robustness"],["The first AI Safety Camp & onwards","Remmelt","2018","blog","LessWrong","www.lesswrong.com/posts/KerENNLyiqQ5ew7Kz/the-first-ai-safety-camp-and-onwards",0,"",""],["Dissolving the Fermi Paradox","Anders Sandberg and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.02404",0,"",""],["Resource-Limited Reflective Oracles","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037555e/resource-limited-reflective-oracles",0,"",""],["The first AI Safety Camp & onwards","Johannes","2018","blog","aisafety.camp","aisafety.camp/2018/06/06/the-first-ai-safety-camp-onwards/",0,"",""],["AISC 1: Research Summaries","Johannes","2018","blog","aisafety.camp","aisafety.camp/2018/06/05/aisc-1-research-summaries/",0,"",""],["Learning to Understand Goal Specifications by Modelling Reward","Dzmitry Bahdanau and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01946",0,"","deception agents policy"],["Measuring and avoiding side effects using relative reachability","Victoria Krakovna","2018","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2018/06/05/measuring-and-avoiding-side-effects-using-relative-reachability/",0,"",""],["Prisoners' Dilemma with Costs to Modeling","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/XjMkPyaPYTf7LrKiT/prisoners-dilemma-with-costs-to-modeling",0,"","theory"],["Relational Deep Reinforcement Learning","Vinicius Zambaldi and 15 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01830",0,"","interpretability agents policy"],["Relational recurrent neural networks","Adam Santoro and 9 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01822",0,"","evals"],["Penalizing side effects using stepwise relative reachability","Victoria Krakovna and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01186",0,"","agents policy"],["Relational inductive bias for physical construction in humans and machines","Jessica B. Hamrick and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01203",0,"","agents policy"],["Relational inductive biases, deep learning, and graph networks","Peter W. Battaglia and 26 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.01261",0,"","interpretability"],["Simplified Poker","Zvi","2018","blog","LessWrong","www.lesswrong.com/posts/i2M3vWPBqyefh3uow/simplified-poker",0,"","theory"],["The Alignment Newsletter #9: 06/04/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/oYbJ2vprujdSP9rcL/the-alignment-newsletter-9-06-04-18",0,"",""],["Between Progress and Potential Impact of AI: the Neglected Dimensions","Fernando Martínez-Plumed and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.00610",0,"",""],["Sufficient Conditions for Idealised Models to Have No Adversarial Examples: a Theoretical and Empirical Study with Bayesian Neural Networks","Yarin Gal and Lewis Smith","2018","paper","arXiv preprint","arxiv.org/abs/1806.00667",0,"","deception robustness"],["May 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/05/31/may-2018-newsletter/",0,"",""],["Special issue on learning for human–robot collaboration","Leonel Rozo and 4 others","2018","report","doi.org","doi.org/10.1007/s10514-018-9756-z",0,"",""],["Agents and Devices: A Relative Definition of Agency","Laurent Orseau and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.12387",0,"","agents"],["Explaining Explanations: An Overview of Interpretability of Machine Learning","Leilani H. Gilpin and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.00069",0,"","interpretability training-data"],["Probabilistically Safe Robot Planning with Confidence-Based Human Predictions","Jaime F. Fisac and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1806.00109",0,"",""],["Managing Loss of Control as Many Militaries Pursue Technological Superiority","Richard Danzig","2018","report","s3.us-east-1.amazonaws.com","s3.us-east-1.amazonaws.com/files.cnas.org/documents/CNASReport-Technology-Roulette-Final.pdf?mtime=20230609105008&focal=none",0,"",""],["Robustness May Be at Odds with Accuracy","Dimitris Tsipras and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.12152",0,"","mechanistic-interpretability robustness"],["To Trust Or Not To Trust A Classifier","Heinrich Jiang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.11783",0,"",""],["Deep Video Portraits","Hyeongwoo Kim and 9 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.11714",0,"",""],["Playing hard exploration games by watching YouTube","Yusuf Aytar and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.11592",0,"","deception agents"],["Variational Inverse Control with Events: A General Framework for Data-Driven Reward Definition","Justin Fu and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.11686",0,"","agents"],["Virtuously Safe Reinforcement Learning","Henrik Aslund and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.11447",0,"","agents policy"],["The Alignment Newsletter #8: 05/28/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/KHLnzFgBtXxJQaDxj/the-alignment-newsletter-8-05-28-18",0,"",""],["The simple picture on AI safety","Alex Flint","2018","blog","LessWrong","www.lesswrong.com/posts/WhNxG4r774bK32GcH/the-simple-picture-on-ai-safety",0,"",""],["Training verified learners with learned verifiers","Krishnamurthy Dvijotham and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.10265",0,"","robustness"],["When is unaligned AI morally valuable?","paulfchristiano","2018","blog","LessWrong","www.lesswrong.com/posts/3kN79EuT27trGexsq/when-is-unaligned-ai-morally-valuable",0,"","robustness"],["Decision theory and zero-sum game theory, NP and PSPACE","jessicata","2018","blog","LessWrong","www.lesswrong.com/posts/KphrG3chfiuFX5Cu6/decision-theory-and-zero-sum-game-theory-np-and-pspace",0,"","theory"],["A Psychopathological Approach to Safety Engineering in AI and AGI","Vahid Behzadan and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08915",0,"","reward-hacking agents"],["Do Better ImageNet Models Transfer Better?","Simon Kornblith and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08974",0,"",""],["Towards the first adversarially robust neural network model on MNIST","Lukas Schott and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.09190",0,"","evals deception robustness"],["How To Solve Moral Conundrums with Computability Theory","Min Baek","2018","paper","arXiv preprint","arxiv.org/abs/1805.08347",0,"",""],["Maximum Causal Tsallis Entropy Imitation Learning","Kyungjae Lee and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08336",0,"","policy"],["Meta-Learning with Hessian-Free Approach in Deep Neural Nets Training","Boyu Chen and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08462",0,"","robustness"],["Verifiable Reinforcement Learning via Policy Extraction","Osbert Bastani and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08328",0,"","policy"],["A Framework and Method for Online Inverse Reinforcement Learning","Saurabh Arora and 2 others","2018","paper","Journal of Autonomous Agents and Multi-Agent Systems, Volume 35,\n  Article number: 4 (2021)","arxiv.org/abs/1805.07871",0,"","agents"],["Constructing Unrestricted Adversarial Examples with Generative Models","Yang Song and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07894",0,"","evals robustness"],["Hierarchical Reinforcement Learning with Hindsight","Andrew Levy and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08180",0,"","agents"],["Imitating Latent Policies from Observation","Ashley D. Edwards and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07914",0,"","policy"],["Learning Safe Policies with Expert Guidance","Jessie Huang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08313",0,"","agents"],["Learning What Information to Give in Partially Observed Domains","Rohan Chitnis and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.08263",0,"","agents"],["The Alignment Newsletter #7: 05/21/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/4ap6WQx52txzsJx6r/the-alignment-newsletter-7-05-21-18",0,"",""],["Constrained Policy Improvement for Safe and Efficient Reinforcement Learning","Elad Sarafian and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07805",0,"","evals policy"],["Task-Agnostic Meta-Learning for Few-shot Learning","Muhammad Abdullah Jamal and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07722",0,"","benchmarks"],["Challenges to Christiano’s capability amplification proposal","Eliezer Yudkowsky","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/S7csET9CgBtpi7sCh/challenges-to-christiano-s-capability-amplification-proposal",0,"","scalable-oversight"],["Challenges to Christiano’s capability amplification proposal","Eliezer Yudkowsky","2018","blog","intelligence.org","intelligence.org/2018/05/19/challenges-to-christianos-capability-amplification-proposal/",0,"","scalable-oversight"],["Solving the Rubik's Cube Without Human Knowledge","Stephen McAleer and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07470",0,"","policy robustness training-data"],["Unsupervised Learning of Neural Networks to Explain Neural Networks","Quanshi Zhang and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.07468",0,"","interpretability evals"],["Lotuses and Loot Boxes","Davidmanheim","2018","blog","LessWrong","www.lesswrong.com/posts/q9F7w6ux26S6JQo3v/lotuses-and-loot-boxes",0,"","reward-hacking goodharts-law"],["The Blessings of Multiple Causes","Yixin Wang and David M. Blei","2018","paper","arXiv preprint","arxiv.org/abs/1805.06826",0,"",""],["Trend in compute used in training for headline AI results","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/trend-in-compute-used-in-training-for-headline-ai-results/",0,"",""],["Feedback-Based Tree Search for Reinforcement Learning","Daniel R. Jiang and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.05935",0,"",""],["RFC: Philosophical Conservatism in AI Alignment Research","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/3r44dhh3uK7s9Pveq/rfc-philosophical-conservatism-in-ai-alignment-research",0,"",""],["The Alignment Newsletter #6: 05/14/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/9pi2XqZP6PYif5wGL/the-alignment-newsletter-6-05-14-18",0,"",""],["Directions and desiderata for AI alignment","Paul Christiano","2018","report","ai-alignment.com","ai-alignment.com/directions-and-desiderata-for-ai-control-b60fca0da8f4",0,"",""],["Thoughts on \"AI safety via debate\"","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/WRy6KNnxwQHc5Ktjc/thoughts-on-ai-safety-via-debate",0,"","debate"],["Automated Mechanism Design via Neural Networks","Weiran Shen and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.03382",0,"",""],["Thoughts on AI Safety via Debate","Vaniver","2018","blog","LessWrong","www.lesswrong.com/posts/h9ZWrrCBgK64pAvxC/thoughts-on-ai-safety-via-debate",0,"","debate"],["Problems integrating decision theory and inverse reinforcement learning","agilecaveman","2018","blog","LessWrong","www.lesswrong.com/posts/nhuMKruWGZH2NCdtH/problems-integrating-decision-theory-and-inverse",0,"","theory"],["The Alignment Newsletter #5: 05/07/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/i7Z9gt6RwzFhokdJ7/the-alignment-newsletter-5-05-07-18",0,"",""],["AI Safety via Debate","ESRogs","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/wo6NsBtn3WJDCeWsx/ai-safety-via-debate",0,"","debate"],["Open question: are minimal circuits daemon-free?","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/nyCHnY7T5PHPLjxmN/open-question-are-minimal-circuits-daemon-free",0,"","mechanistic-interpretability"],["AGI Safety Literature Review (Everitt, Lea & Hutter 2018)","Kaj_Sotala","2018","blog","LessWrong","www.lesswrong.com/posts/RAZNGucjxAcKHimcS/agi-safety-literature-review-everitt-lea-and-hutter-2018",0,"",""],["Dynamic Control Flow in Large-Scale Machine Learning","Yuan Yu and 14 others","2018","paper","EuroSys 2018: Thirteenth EuroSys Conference, April 23-26, 2018,\n  Porto, Portugal. ACM, New York, NY, USA","arxiv.org/abs/1805.01772",0,"","evals"],["Everything I ever needed to know, I learned from World of Warcraft: Goodhart’s law","Said Achmiz","2018","blog","LessWrong","www.lesswrong.com/posts/GxW8ef8tH4yX6KMrf/everything-i-ever-needed-to-know-i-learned-from-world-of-2",0,"","goodharts-law"],["Rigging is a form of wireheading","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/b8HauRWrjBdnKEwM5/rigging-is-a-form-of-wireheading",0,"","reward-hacking"],["Reinforcement Learning and Control as Probabilistic Inference: Tutorial and Review","Sergey Levine","2018","paper","arXiv preprint","arxiv.org/abs/1805.00909",0,"","policy"],["Soon: a weekly AI Safety prerequisites module on LessWrong","anonymous","2018","blog","LessWrong","www.lesswrong.com/posts/vi48CMtZL8ZkRpuad/soon-a-weekly-ai-safety-prerequisites-module-on-lesswrong",0,"",""],["The Alignment Newsletter #4: 04/30/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/27fiu3ur57HmfEhTv/the-alignment-newsletter-4-04-30-18",0,"",""],["Issues with Iterated Distillation and Amplification","Luca Rade","2018","report","medium.com","medium.com/@lucarade/issues-with-iterated-distillation-and-amplification-5aa01ab37173",0,"","scalable-oversight"],["Levels of AI Self-Improvement","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/os7N7nJoezWKQnnuW/levels-of-ai-self-improvement",0,"","forecasting"],["A Logic of Agent Organizations","Virginia Dignum and Frank Dignum","2018","paper","Logic Journal of the IGPL, vol. 20, no. 1, pp. 283-316, Feb. 2012","arxiv.org/abs/1804.10817",0,"","agents"],["Double Cruxing the AI Foom debate","agilecaveman","2018","blog","LessWrong","www.lesswrong.com/posts/b2MnFM8DWDaPhxBoK/double-cruxing-the-ai-foom-debate",0,"","forecasting"],["Noticing the Taste of Lotus","Valentine","2018","blog","LessWrong","www.lesswrong.com/posts/KwdcMts8P8hacqwrX/noticing-the-taste-of-lotus",0,"","goodharts-law"],["Reward Learning from Narrated Demonstrations","Hsiao-Yu Fish Tung and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.10692",0,"",""],["Goertzel’s GOLEM implements evidential decision theory applied to policy choice","Caspar Oesterheld","2018","report","casparoesterheld.com","casparoesterheld.com/2018/04/26/goertzels-golem-implements-evidential-decision-theory-applied-to-policy-choice/",0,"","policy theory"],["The Best of Both Worlds: Combining Recent Advances in Neural Machine Translation","Mia Xu Chen and 11 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.09849",0,"","benchmarks"],["No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","Xin Wang and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.09160",0,"","evals policy robustness"],["Realistic Evaluation of Deep Semi-Supervised Learning Algorithms","Avital Oliver and 4 others","2018","paper","NeurIPS 2018 Proceedings","arxiv.org/abs/1804.09170",0,"","evals benchmarks"],["The Alignment Newsletter #3: 04/23/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/YvbbkPYH77xhdqvKt/the-alignment-newsletter-3-04-23-18",0,"",""],["Zero-Shot Visual Imitation","Deepak Pathak and 9 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.08606",0,"","evals agents policy"],["Preventing Side-effects in Gridworlds","Gavin Leech and 3 others","2018","report","gleech.org","www.gleech.org/grids",0,"",""],["Deep Probabilistic Programming Languages: A Qualitative Study","Guillaume Baudart and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.06458",0,"",""],["Understanding Iterated Distillation and Amplification: Claims and Oversight","William_S","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/yxzrKb2vFXRkwndQ4/understanding-iterated-distillation-and-amplification-claims",0,"","scalable-oversight"],["Heuristic Approaches for Goal Recognition in Incomplete Domain Models","Ramon Fraga Pereira and Felipe Meneguzzi","2018","paper","arXiv preprint","arxiv.org/abs/1804.05917",0,"",""],["On Gradient-Based Learning in Continuous Games","Eric Mazumdar and 2 others","2018","paper","SIAM Journal on Mathematics of Data Science 2020 2:1, 103-131","arxiv.org/abs/1804.05464",0,"","benchmarks deception agents policy"],["The Alignment Newsletter #2: 04/16/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/fZmMLCnZmMF9xgrs5/the-alignment-newsletter-2-04-16-18",0,"",""],["Adversarial Attacks Against Medical Deep Learning Systems","Samuel G. Finlayson and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.05296",0,"","robustness"],["Implicit extortion","paulfchristiano","2018","blog","LessWrong","www.lesswrong.com/posts/XfRB26FqXFrTh83Pf/implicit-extortion",0,"","theory"],["Implicit extortion","Paul Christiano","2018","report","ai-alignment.com","ai-alignment.com/implicit-extortion-3c80c45af1e3",0,"",""],["Quantilal control for finite MDPs","Vanessa Kosoy","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375556/quantilal-control-for-finite-mdps",0,"",""],["Capsules for Object Segmentation","Rodney LaLonde and Ulas Bagci","2018","paper","arXiv preprint","arxiv.org/abs/1804.04241",0,"",""],["Emergent Communication through Negotiation","Kris Cao and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.03980",0,"","agents"],["April 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/04/10/april-2018-newsletter/",0,"",""],["The limits of corrigibility","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/T5ZyNq3fzN59aQG5y/the-limits-of-corrigibility",0,"",""],["First Experiments with a Flexible Infrastructure for Normative Reasoning","Christoph Benzmüller and Xavier Parent","2018","paper","arXiv preprint","arxiv.org/abs/1804.02929",0,"","governance"],["Large scale distributed neural network training through online distillation","Rohan Anil and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.03235",0,"",""],["The Alignment Newsletter #1: 04/09/18","Rohin Shah","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/CsMQ7zsprBqWaeSvk/the-alignment-newsletter-1-04-09-18",0,"",""],["Two guarantees","Paul Christiano","2018","report","ai-alignment.com","ai-alignment.com/two-guarantees-c4c03a6b434f",0,"","scalable-oversight policy robustness"],["Fortified Networks: Improving the Robustness of Deep Networks by Modeling the Manifold of Hidden Representations","Alex Lamb and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.02485",0,"","evals robustness"],["Programmatically Interpretable Reinforcement Learning","Abhinav Verma and 4 others","2018","paper","PMLR 80:5045-5054","arxiv.org/abs/1804.02477",0,"","interpretability evals agents policy robustness"],["Promising research projects","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/promising-research-projects/",0,"",""],["The tyranny of the god scenario","Michael Wulfsohn","2018","blog","aiimpacts.org","aiimpacts.org/the-tyranny-of-the-god-scenario/",0,"",""],["Poison Frogs! Targeted Clean-Label Poisoning Attacks on Neural Networks","Ali Shafahi and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.00792",0,"","training-data"],["Specification gaming examples in AI","Vika","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/AanbbjYr5zckMKde7/specification-gaming-examples-in-ai-1",0,"","specification-gaming goodharts-law"],["Universal Planning Networks","Aravind Srinivas and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.00645",0,"",""],["2018 research plans and predictions","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/03/31/2018-research-plans/",0,"",""],["Artificial Intelligence and its Role in Near Future","Jahanzaib Shabbir and Tarique Anwer","2018","paper","arXiv preprint","arxiv.org/abs/1804.01396",0,"","agents"],["Can corrigibility be learned safely?","Wei Dai","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/o22kP33tumooBtia3/can-corrigibility-be-learned-safely",0,"",""],["Corrigible but misaligned: a superintelligent messiah","zhukeepa","2018","blog","LessWrong","www.lesswrong.com/posts/mSYR46GZZPMmX7q93/corrigible-but-misaligned-a-superintelligent-messiah",0,"",""],["My take on agent foundations: formalizing metaphilosophical competence","zhukeepa","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/xCpuSfT5Lt6kkR3po/my-take-on-agent-foundations-formalizing-metaphilosophical",0,"","agents theory"],["Specification gaming examples in AI","Victoria Krakovna","2018","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2018/04/02/specification-gaming-examples-in-ai/",0,"","specification-gaming"],["Adversarial Attacks and Defences Competition","Alexey Kurakin and 22 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.00097",0,"",""],["Iterative Learning with Open-set Noisy Labels","Yisen Wang and 6 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.00092",0,"","monitoring"],["Meta-Learning Update Rules for Unsupervised Representation Learning","Luke Metz and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1804.00222",0,"","robustness"],["Opportunities for individual donors in AI safety","Alex Flint","2018","blog","LessWrong","www.lesswrong.com/posts/cXbXR7QCqWvmPzjki/opportunities-for-individual-donors-in-ai-safety",0,"",""],["Three wagers for multiverse-wide superrationality","Johannes Treutlein","2018","report","casparoesterheld.com","casparoesterheld.com/2018/03/31/three-wagers-for-multiverse-wide-superrationality/",0,"",""],["Brain wiring: The long and short of it","Tegan McCaslin","2018","blog","aiimpacts.org","aiimpacts.org/brain-wiring-the-long-and-short-of-it/",0,"",""],["Resolving human values, completely and adequately","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/Y2LhX3925RodndwpC/resolving-human-values-completely-and-adequately",0,"",""],["Reward hacking and Goodhart’s law by evolutionary algorithms","Jan_Kulveit","2018","blog","LessWrong","www.lesswrong.com/posts/CbQBJaZCrGMJEBz8g/reward-hacking-and-goodhart-s-law-by-evolutionary-algorithms",0,"","reward-hacking goodharts-law"],["Transmitting fibers in the brain: Total length and distribution of lengths","Tegan McCaslin","2018","blog","aiimpacts.org","aiimpacts.org/transmitting-fibers-in-the-brain-total-length-and-distribution-of-lengths/",0,"",""],["Autonomous Intelligent Cyber-defense Agent (AICA) Reference Architecture. Release 2.0","Alexander Kott and 11 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.10664",0,"","agents"],["New paper: “Categorizing variants of Goodhart’s Law”","Scott Garrabrant","2018","blog","intelligence.org","intelligence.org/2018/03/27/categorizing-goodhart/",0,"","goodharts-law"],["Evaluating Existing Approaches to AGI Alignment","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/RMhs2fXtK5hLAjDQv/evaluating-existing-approaches-to-agi-alignment",0,"","evals"],["My Thoughts on Takeoff Speeds","tristanm","2018","blog","LessWrong","www.lesswrong.com/posts/LaT6rexiNx6MW74Fn/my-thoughts-on-takeoff-speeds",0,"","forecasting"],["Non-Adversarial Goodhart and AI Risks","Davidmanheim","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/iK2F9QDZvwWinsBYB/non-adversarial-goodhart-and-ai-risks",0,"","goodharts-law"],["World Models","David Ha and Jürgen Schmidhuber","2018","report","worldmodels.github.io","worldmodels.github.io/",0,"",""],["March 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/03/25/march-2018-newsletter/",0,"",""],["Computational Power and the Social Impact of Artificial Intelligence","Tim Hwang","2018","paper","arXiv preprint","arxiv.org/abs/1803.08971",0,"","deception"],["Idea: Open Access AI Safety Journal","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/a65sFvymnoLkBnE8n/idea-open-access-ai-safety-journal",0,"",""],["Learning-based Model Predictive Control for Safe Exploration","Torsten Koller and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.08287",0,"",""],["An Untrollable Mathematician Illustrated","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/CvKnhXTu9BPcdKE4W/an-untrollable-mathematician-illustrated",0,"",""],["Generating Multi-Agent Trajectories using Programmatic Weak Supervision","Eric Zhan and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.07612",0,"","interpretability evals agents"],["AI Alignment Prize: Super-Boxing","X4vier","2018","blog","LessWrong","www.lesswrong.com/posts/DTv3jpro99KwdkHRE/ai-alignment-prize-super-boxing",0,"",""],["Deciphering China's AI Dream","Qiaochu_Yuan","2018","blog","LessWrong","www.lesswrong.com/posts/D52m6f3F76jp8scYA/deciphering-china-s-ai-dream",0,"","governance"],["Distributed Cooperation","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037554e/distributed-cooperation",0,"",""],["Is the Star Trek Federation really incapable of building AI?","Kaj_Sotala","2018","blog","LessWrong","www.lesswrong.com/posts/7N7JGyTmX5Gnjhfrk/is-the-star-trek-federation-really-incapable-of-building-ai",0,"",""],["A Dual Approach to Scalable Verification of Deep Networks","Krishnamurthy and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.06567",0,"","robustness"],["Adversarial Logit Pairing","Harini Kannan and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.06373",0,"","robustness"],["Jan Leike on how to become a machine learning alignment researcher","Robert Wiblin and 2 others","2018","report","80000hours.org","80000hours.org/podcast/episodes/jan-leike-ml-alignment/",0,"",""],["Neural Network Quine","Oscar Chang and Hod Lipson","2018","paper","arXiv preprint","arxiv.org/abs/1803.05859",0,"",""],["Using lying to detect human values","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/pQz97SLCRMwHs6BzF/using-lying-to-detect-human-values",0,"","deception"],["Active Reinforcement Learning with Monte-Carlo Tree Search","Sebastian Schulze and Owain Evans","2018","paper","arXiv preprint","arxiv.org/abs/1803.04926",0,"","agents"],["Categorizing Variants of Goodhart's Law","David Manheim and Scott Garrabrant","2018","paper","arXiv preprint","arxiv.org/abs/1803.04585",0,"","goodharts-law governance policy"],["Deep k-Nearest Neighbors: Towards Confident, Interpretable and Robust Deep Learning","Nicolas Papernot and Patrick McDaniel","2018","paper","arXiv preprint","arxiv.org/abs/1803.04765",0,"","interpretability evals robustness training-data"],["Fractal AI: A fragile theory of intelligence","Sergio Hernandez Cerezo and Guillem Duran Ballester","2018","paper","arXiv preprint","arxiv.org/abs/1803.05049",0,"","benchmarks agents"],["AI Alignment Prize: Round 2 due March 31, 2018","Zvi","2018","blog","LessWrong","www.lesswrong.com/posts/rgWLPuQAxwoikpRu5/ai-alignment-prize-round-2-due-march-31-2018",0,"",""],["Opportunities for individual donors in AI safety","alexflint","2018","blog","EA Forum","forum.effectivealtruism.org/posts/HqatEhdEb42vhSo7B/opportunities-for-individual-donors-in-ai-safety",0,"",""],["Brains and backprop: a key timeline crux","jacobjacob","2018","blog","LessWrong","www.lesswrong.com/posts/QWyYcjrXASQuRHqC5/brains-and-backprop-a-key-timeline-crux",0,"","forecasting"],["Institutional Metaphors for Designing Large-Scale Distributed AI versus AI Techniques for Running Institutions","Alexander Boer and Giovanni Sileno","2018","paper","arXiv preprint","arxiv.org/abs/1803.03407",0,"","agents"],["The Challenge of Crafting Intelligible Intelligence","Daniel S. Weld and Gagan Bansal","2018","paper","arXiv preprint","arxiv.org/abs/1803.04263",0,"","interpretability"],["The Surprising Creativity of Digital Evolution: A Collection of Anecdotes from the Evolutionary Computation and Artificial Life Research Communities","Joel Lehman and 39 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.03453",0,"",""],["Prize for probable problems","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/SqcPWvvJJwwgZb6aH/prize-for-probable-problems",0,"","scalable-oversight"],["SentRNA: Improving computational RNA design by incorporating a prior of human design strategies","Jade Shi and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1803.03146",0,"","agents"],["A Brandom-ian view of Reinforcement Learning towards strong-AI","Atrisha Sarkar","2018","paper","arXiv preprint","arxiv.org/abs/1803.02912",0,"",""],["Value Alignment, Fair Play, and the Rights of Service Robots","Daniel Estrada","2018","paper","ACM/AIES 2018","arxiv.org/abs/1803.02852",0,"",""],["The Building Blocks of Interpretability","Chris Olah and 6 others","2018","report","Distill","distill.pub/2018/building-blocks",0,"","interpretability"],["Iterated Distillation and Amplification","Ajeya Cotra","2018","report","ai-alignment.com","ai-alignment.com/iterated-distillation-and-amplification-157debfd1616",0,"","scalable-oversight"],["Takeoff Speed: Simple Asymptotics in a Toy Model.","Aaron Roth","2018","blog","LessWrong","www.lesswrong.com/posts/7MK6HSn2pbAJrbfiG/takeoff-speed-simple-asymptotics-in-a-toy-model",0,"","forecasting"],["AI impacts and Paul Christiano on takeoff speeds","Crosspost","2018","blog","EA Forum","forum.effectivealtruism.org/posts/zFiGfWbZGgT8sGwRC/ai-impacts-and-paul-christiano-on-takeoff-speeds",0,"","forecasting"],["Explainable Robotic Systems","Maartje M.A. de Graaf and 3 others","2018","report","dl.acm.org","dl.acm.org/citation.cfm?doid=3173386.3173568",0,"",""],["Human-aligned artificial intelligence is a multiobjective problem","Peter Vamplew and 4 others","2018","report","doi.org","doi.org/10.1007/s10676-017-9440-6",0,"",""],["Quick Nate/Eliezer comments on discontinuity","Rob Bensinger","2018","blog","LessWrong","www.lesswrong.com/posts/X5zmEvFQunxiEcxHn/quick-nate-eliezer-comments-on-discontinuity",0,"","forecasting"],["Sam Harris and Eliezer Yudkowsky on “AI: Racing Toward the Brink”","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/02/28/sam-harris-and-eliezer-yudkowsky/",0,"",""],["Beyond algorithmic equivalence: algorithmic noise","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/meG3Pai2YeRYcwPwS/beyond-algorithmic-equivalence-algorithmic-noise",0,"",""],["Beyond algorithmic equivalence: self-modelling","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/kmLP3bTnBhc22DnqY/beyond-algorithmic-equivalence-self-modelling",0,"",""],["TDT for Humans","alkjash","2018","blog","LessWrong","www.lesswrong.com/posts/4Kye4kkKwn6DCahKy/tdt-for-humans",0,"","agents theory"],["Antifragility for Intelligent Autonomous Systems","Anusha Mujumdar and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.09159",0,"","agents"],["February 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/02/25/february-2018-newsletter/",0,"",""],["Learning from Physical Human Corrections, One Feature at a Time","Andrea Bajcsy and 3 others","2018","report","dl.acm.org","dl.acm.org/citation.cfm?doid=3171221.3171267",0,"",""],["More on the Linear Utility Hypothesis and the Leverage Prior","AlexMennen","2018","blog","LessWrong","www.lesswrong.com/posts/mBFqG3xjYazsPiZkH/more-on-the-linear-utility-hypothesis-and-the-leverage-prior",0,"","theory"],["Walkthrough of 'Formalizing Convergent Instrumental Goals'","TurnTrout","2018","blog","LessWrong","www.lesswrong.com/posts/KXMqckn9avvY4Zo9W/walkthrough-of-formalizing-convergent-instrumental-goals",0,"","instrumental-convergence"],["Will AI See Sudden Progress?","KatjaGrace","2018","blog","LessWrong","www.lesswrong.com/posts/AJtfNyBsum6ZzWxKR/will-ai-see-sudden-progress",0,"","forecasting"],["Arguments about fast takeoff","paulfchristiano","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/AfGmsjGPXN97kNp57/arguments-about-fast-takeoff",0,"","forecasting"],["Self-regulation of safety in AI research","Gordon Seidoh Worley","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/Z8BWP6CEQuARcbNZu/self-regulation-of-safety-in-ai-research",0,"","governance"],["The abruptness of nuclear weapons","paulfchristiano","2018","blog","LessWrong","www.lesswrong.com/posts/y5eapqjYYku8Wt9wn/the-abruptness-of-nuclear-weapons",0,"","forecasting"],["Will AI see sudden progress?","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/will-ai-see-sudden-progress/",0,"",""],["Takeoff speeds","paulfchristiano","2018","report","sideways-view.com","sideways-view.com/2018/02/24/takeoff-speeds/",0,"","forecasting"],["June 2012: 0/33 Turing Award winners predict computers beating humans at go within next 10 years.","betterthanwell","2018","blog","LessWrong","www.lesswrong.com/posts/cEhv4yd6GYgz66LgK/june-2012-0-33-turing-award-winners-predict-computers",0,"","forecasting"],["Likelihood of discontinuous progress around the development of AGI","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/likelihood-of-discontinuous-progress-around-the-development-of-agi/",0,"","robustness"],["Don't Condition on no Catastrophes","Scott Garrabrant","2018","blog","LessWrong","www.lesswrong.com/posts/8NBbq7xhyDXoDWM8e/don-t-condition-on-no-catastrophes",0,"","forecasting"],["Machine Theory of Mind","Neil C. Rabinowitz and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.07740",0,"","interpretability agents"],["Manipulating and Measuring Model Interpretability","Forough Poursabzi-Sangdeh and 4 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.07810",0,"","interpretability"],["Robustness to Scale","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/bBdfbWfWxHN9Chjcq/robustness-to-scale",0,"","agents robustness"],["The Malicious Use of Artificial Intelligence: Forecasting, Prevention, and Mitigation","Miles Brundage and 25 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.07228",0,"","forecasting"],["Using surrogate goals to deflect threats","Tobias Baumann","2018","report","longtermrisk.org","longtermrisk.org/using-surrogate-goals-deflect-threats/",0,"",""],["Why I prioritize moral circle expansion over reducing extinction risk through artificial intelligence alignment","Jacy","2018","blog","EA Forum","forum.effectivealtruism.org/posts/BY8gXSpGijypbGitT/why-i-prioritize-moral-circle-expansion-over-reducing",0,"",""],["Electrical efficiency of computing","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/electrical-efficiency-of-computing/",0,"",""],["Learning Data-Driven Objectives to Optimize Interactive Systems","Ziming Li and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.06306",0,"",""],["Nordhaus hardware price performance dataset","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/nordhaus-hardware-price-performance-dataset/",0,"",""],["Toward a New Technical Explanation of Technical Explanation","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/tKwJQbo6SfWF2ifKh/toward-a-new-technical-explanation-of-technical-explanation",0,"","theory"],["Adversarial Risk and the Dangers of Evaluating Against Weak Attacks","Jonathan Uesato and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.05666",0,"","evals robustness"],["The law of effect, randomization and Newcomb’s problem","Caspar Oesterheld","2018","report","casparoesterheld.com","casparoesterheld.com/2018/02/15/the-law-of-effect-randomization-and-newcombs-problem/",0,"",""],["Two Types of Updatelessness","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/pneKTZG9KqnSe2RdQ/two-types-of-updatelessness",0,"","theory"],["2018 price of performance by Tensor Processing Units","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/2018-price-of-performance-by-tensor-processing-units/",0,"",""],["Examples of AI systems producing unconventional solutions","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/examples-of-ai-systems-producing-unconventional-solutions/",0,"",""],["Some conceptual highlights from “Disjunctive Scenarios of Catastrophic AI Risk”","Kaj_Sotala","2018","blog","LessWrong","www.lesswrong.com/posts/8uJ3n3hu8pLXC4YNE/some-conceptual-highlights-from-disjunctive-scenarios-of-1",0,"","forecasting"],["Historic trends in altitude","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/discontinuity-in-altitude-records/",0,"",""],["More Robust Doubly Robust Off-policy Evaluation","Mehrdad Farajtabar and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.03493",0,"","evals benchmarks policy"],["Knowledge is Freedom","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/b3Bt9Cz4hEtR26ANX/knowledge-is-freedom",0,"","theory"],["Stable Pointers to Value II: Environmental Goals","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/wujPGixayiZSMYfm6/stable-pointers-to-value-ii-environmental-goals",0,"",""],["Goal Inference Improves Objective and Perceived Performance in Human-Robot Collaboration","Chang Liu and 6 others","2018","paper","C. Liu, J. Hamrick, J. Fisac, A. Dragan, J. K. Hedrick, S. Sastry,\n  T. Griffiths. \"Goal Inference Improves Objective and Perceived Performance in\n  Human-Robot Collaboration\". Autonomous Agents and Multiagent Systems (AAMAS),\n  2016","arxiv.org/abs/1802.01780",0,"","evals agents robustness"],["Shared Autonomy via Deep Reinforcement Learning","Siddharth Reddy and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.01744",0,"","deception agents policy"],["UDT as a Nash Equilibrium","cousin_it","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/6HmaGnXd4EJfpfait/udt-as-a-nash-equilibrium",0,"","theory"],["First-order Adversarial Vulnerability of Neural Networks and Input Dimension","Carl-Johann Simon-Gabriel and 4 others","2018","paper","Proceedings of ICML 2019","arxiv.org/abs/1802.01421",0,"","robustness"],["Learning from Richer Human Guidance: Augmenting Comparison-Based Learning with Feature Queries","Chandrayee Basu and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.01604",0,"",""],["Factorio, Accelerando, Empathizing with Empires and Moderate Takeoffs","Raemon","2018","blog","LessWrong","www.lesswrong.com/posts/RHurATLtM7S5JWe9v/factorio-accelerando-empathizing-with-empires-and-moderate",0,"","forecasting"],["Logical counterfactuals and differential privacy","Nisan","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375532/logical-counterfactuals-and-differential-privacy",0,"",""],["AI Safety Research Camp - Project Proposal","David_Kristoffersson","2018","blog","LessWrong","www.lesswrong.com/posts/KgFrtaajjfSnBSZoH/ai-safety-research-camp-project-proposal",0,"",""],["Techniques for optimizing worst-case performance","Paul Christiano","2018","report","ai-alignment.com","ai-alignment.com/techniques-for-optimizing-worst-case-performance-39eafec74b99",0,"",""],["The Utility of Human Atoms for the Paperclip Maximizer","avturchin","2018","blog","LessWrong","www.lesswrong.com/posts/BCkdLTJMn9zZuAzAh/the-utility-of-human-atoms-for-the-paperclip-maximizer",0,"","instrumental-convergence"],["Bias in AI: How we Build Fair AI Systems and Less-Biased Humans","Anonymous","2018","report","ibm.com","www.ibm.com/policy/bias-in-ai/",0,"",""],["Obfuscated Gradients Give a False Sense of Security: Circumventing Defenses to Adversarial Examples","Anish Athalye and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1802.00420",0,"","deception robustness"],["The Malicious Use of Artificial Intelligence: Forecasting, Prevention, and Mitigation (2018). Brundage and Avin et al.","Miles Bridge and 2 others","2018","report","docs.google.com","docs.google.com/document/d/e/2PACX-1vQzbSybtXtYzORLqGhdRYXUqiFsaEOvftMSnhVgJ-jRh6plwkzzJXoQ-sKtej3HW_0pzWTFY7-1eoGf/pub",0,"","forecasting"],["Epiphenomenal Oracles Ignore Holes in the Box","SilentCal","2018","blog","LessWrong","www.lesswrong.com/posts/q5qoG7gXuntKgBNoR/epiphenomenal-oracles-ignore-holes-in-the-box",0,"",""],["Sources of intuitions and data on AGI","Scott Garrabrant","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/BibDWWeo37pzuZCmL/sources-of-intuitions-and-data-on-agi",0,"",""],["January 2018 Newsletter","Rob Bensinger","2018","blog","intelligence.org","intelligence.org/2018/01/28/january-2018-newsletter/",0,"",""],["Against Instrumental Convergence","zulupineapple","2018","blog","LessWrong","www.lesswrong.com/posts/28kcq8D4aCWeDKbBp/against-instrumental-convergence",0,"","instrumental-convergence"],["Is there a tradeoff between immediate and longer-term AI safety efforts?","Victoria Krakovna","2018","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2018/01/27/is-there-a-tradeoff-between-safety-concerns-about-current-and-future-ai-systems/",0,"",""],["Strategy Nonconvexity Induced by a Choice of Potential Oracles","Diffractor","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037553c/strategy-nonconvexity-induced-by-a-choice-of-potential-oracles",0,"",""],["Safe Exploration in Continuous Action Spaces","Gal Dalal and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1801.08757",0,"","agents policy"],["Space races: Settling the universe Fast","Anders Sandberg","2018","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/space-races-settling.pdf",0,"",""],["An Untrollable Mathematician","abramdemski","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375533/an-untrollable-mathematician",0,"",""],["AI alignment prize winners and next round [link]","RyanCarey","2018","blog","EA Forum","forum.effectivealtruism.org/posts/HnxQF6kkLuyiSZjhN/ai-alignment-prize-winners-and-next-round-link",0,"",""],["2015 FLOPS prices","Katja Grace","2018","blog","aiimpacts.org","aiimpacts.org/2015-flops-prices/",0,"",""],["A model I use when making plans to reduce AI x-risk","Ben Pace","2018","blog","LessWrong","www.lesswrong.com/posts/XFpDTCHZZ4wpMT8PZ/a-model-i-use-when-making-plans-to-reduce-ai-x-risk",0,"","forecasting"],["Beware of black boxes in AI alignment research","cousin_it","2018","blog","AI Alignment Forum","www.alignmentforum.org/posts/DNKTmmNZr5M2uCZLz/beware-of-black-boxes-in-ai-alignment-research",0,"",""],["Symmetric Decomposition of Asymmetric Games","Karl Tuyls and 8 others","2018","report","nature.com","www.nature.com/articles/s41598-018-19194-4",0,"",""],["Towards an Integrated Assessment of Global Catastrophic Risk","Seth Baum and Anthony M Barrett","2018","report","sethbaum.com","sethbaum.com/ac/2017_Integrated.pdf",0,"",""],["Announcement: AI alignment prize winners and next round","cousin_it","2018","blog","LessWrong","www.lesswrong.com/posts/4WbNGQMvuFtY3So7s/announcement-ai-alignment-prize-winners-and-next-round",0,"",""],["Counterfactual equivalence for POMDPs, and underlying deterministic environments","Stuart Armstrong","2018","paper","arXiv preprint","arxiv.org/abs/1801.03737",0,"","deception agents"],["Fundraising success!","Malo Bourgon","2018","blog","intelligence.org","intelligence.org/2018/01/10/fundraising-success/",0,"",""],["Spatially Transformed Adversarial Examples","Chaowei Xiao and 5 others","2018","paper","arXiv preprint","arxiv.org/abs/1801.02612",0,"","robustness"],["2017-18 New Year review","Victoria Krakovna","2018","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2018/01/07/2017-18-new-year-review/",0,"",""],["Global online debate on the governance of AI","CarolineJ","2018","blog","LessWrong","www.lesswrong.com/posts/ny5soHNLpjMoMHTZa/global-online-debate-on-the-governance-of-ai",0,"","governance"],["Have you felt exiert yet?","Stuart_Armstrong","2018","blog","LessWrong","www.lesswrong.com/posts/qP3s89RAcdYy2LN2K/have-you-felt-exiert-yet",0,"",""],["Papers for 2017","Kaj_Sotala","2018","blog","LessWrong","www.lesswrong.com/posts/beSGFi2Z9uidL5rrN/papers-for-2017",0,"","forecasting"],["A Rational Reinterpretation of Dual-Process Theories","Smitha Milli and 2 others","2018","report","rgdoi.net","rgdoi.net/10.13140/RG.2.2.14956.46722/1",0,"",""],["Adversarial Examples Are a Natural Consequence of Test Error in Noise","Nicolas Ford and 3 others","2018","paper","arXiv preprint","arxiv.org/abs/1901.10513",0,"","robustness"],["AGI Safety Literature Review","Tom Everitt and 2 others","2018","paper","arXiv preprint","arxiv.org/abs/1805.01109",0,"",""],["AI governance research agenda","Allan Dafoe","2018","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/GovAIAgenda.pdf",0,"","governance"],["Artificial General Intelligence: Coordination & Great Powers","Allison Duettmann and 9 others","2018","report","foresight.org","foresight.org/wp-content/uploads/2018/11/AGI-Coordination-Geat-Powers-Report.pdf",0,"",""],["Artificial General Intelligence: Coordination and Great Powers","Allison Duettman and 9 others","2018","report","fsone-bb4c.kxcdn.com","fsone-bb4c.kxcdn.com/wp-content/uploads/2018/11/AGI-Coordination-Geat-Powers-Report.pdf",0,"",""],["Certified Defenses against Adversarial Examples","Aditi Raghunathan and Jacob Steinhardt & Percy Liang","2018","paper","arXiv preprint","arxiv.org/abs/1801.09344",0,"","robustness"],["Countering Superintelligence Misinformation","Seth Baum","2018","report","gcrinstitute.org","gcrinstitute.org/countering-superintelligence-misinformation/",0,"",""],["Deciphering China’s AI dream","Jeffrey Ding","2018","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Deciphering_Chinas_AI-Dream.pdf",0,"",""],["GLOBAL POLITICS AND THE GOVERNANCE OF ARTIFICIAL INTELLIGENCE","Allan Dafoe and Journal of International Affairs","2018","report","jstor.org","www.jstor.org/stable/26588347",0,"","governance"],["Hacking the brain: dimensions of cognitive enhancement","Martin Dresler and 7 others","2018","report","ncbi.nlm.nih.gov","www.ncbi.nlm.nih.gov/pmc/articles/PMC6429408/pdf/cn8b00571.pdf",0,"",""],["How rapidly are GPUs improving in price performance?","Baeo Maltinsky","2018","report","mediangroup.org","mediangroup.org/gpu.html",0,"",""],["ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness","Robert Geirhos","2018","paper","arXiv preprint","arxiv.org/abs/1811.12231",0,"","robustness"],["Insight-based AI timelines model","Baeo Maltinsky","2018","report","mediangroup.org","mediangroup.org/insights",0,"","forecasting"],["Introduction to STAMP","Nancy G. Leveson","2018","report","psas.scripts.mit.edu","psas.scripts.mit.edu/home/wp-content/uploads/2020/07/STAMP-Tutorial.pdf",0,"",""],["Motivating the Rules of the Game for Adversarial Example Research","Justin Gilmer","2018","paper","arXiv preprint","arxiv.org/abs/1807.06732",0,"","robustness"],["Negotiable Reinforcement Learning for Pareto Optimal Sequential Decision-Making","Nishant Desai and 2 others","2018","report","papers.nips.cc","papers.nips.cc/paper/7721-negotiable-reinforcement-learning-for-pareto-optimal-sequential-decision-making.pdf",0,"",""],["Occam's razor is insufficient to infer the preferences of irrational agents","Stuart Armstrong and Sören Mindermann","2018","report","proceedings.neurips.cc","proceedings.neurips.cc/paper_files/paper/2018/file/d89a66c7c80a29b1bdbab0f2a1a94af8-Paper.pdf",0,"","agents"],["On Calibration of Modern Neural Networks","Chuan Guo and 7 others","2018","paper","arXiv preprint","arxiv.org/abs/1706.04599",0,"",""],["Predicting Human Deliberative Judgments with Machine Learning","Owain Evans and 6 others","2018","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/predicting-judgments-tr2018.pdf",0,"",""],["Public Policy and Superintelligent AI: A Vector Field Approach","Nick Bostrom and 2 others","2018","report","nickbostrom.com","nickbostrom.com/papers/aipolicy.pdf",0,"","policy"],["Reconciliation between factions focused on near-term and long-term artificial intelligence","Seth D. Baum","2018","report","papers.ssrn.com","papers.ssrn.com/sol3/papers.cfm?abstract_id=2976444",0,"",""],["Superintelligence skepticism as a political tool","Seth Baum","2018","report","mdpi.com","www.mdpi.com/2078-2489/9/9/209",0,"",""],["The Brain and Computation","Baeo Maltinsky","2018","report","mediangroup.org","mediangroup.org/brain1.html",0,"",""],["The new weapons of mass destruction?","Ronald Arkin and 2 others","2018","report","the-security-times.com","www.the-security-times.com/wp-content/uploads/2018/02/ST_Feb2018_Doppel-2.pdf",0,"",""],["The State of Research in Existential Risk","Seán Ó hÉigeartaigh","2018","report","risksciences.ucla.edu","www.risksciences.ucla.edu/news-events/2018/1/2/proceedings-of-the-first-international-colloquium-on-catastrophic-and-existential-risk",0,"",""],["The vulnerable world hypothesis","Nick Bostrom","2018","report","nickbostrom.com","nickbostrom.com/papers/vulnerable.pdf",0,"","governance"],["Toward A Working Theory of Mind","Miya Perry","2018","report","mediangroup.org","mediangroup.org/docs/toward_a_working_theory_of_mind.pdf",0,"",""],["Where Do You Think You're Going?: Inferring Beliefs about Dynamics from Behavior","Sid Reddy and 2 others","2018","report","papers.nips.cc","papers.nips.cc/paper/7419-where-do-you-think-youre-going-inferring-beliefs-about-dynamics-from-behavior.pdf",0,"",""],["Goodhart Taxonomy","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/EbFABnst8LsidYs5Y/goodhart-taxonomy",0,"","goodharts-law"],["NIPS 2017 Report","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/12/30/nips-2017-report/",0,"",""],["The Three Levels of Goodhart's Curse","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754b2/the-three-levels-of-goodhart-s-curse",0,"","goodharts-law"],["Effect of marginal hardware on artificial general intelligence","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/effect-of-marginal-hardware-on-artificial-general-intelligence/",0,"",""],["Artificial Intelligence in Life Extension: from Deep Learning to Superintelligence","Mikhail Batin and 4 others","2017","report","informatica.si","www.informatica.si/index.php/informatica/article/download/1797/1104",0,"",""],["Conceptual-Linguistic Superintelligence","David J. Jilk","2017","report","informatica.si","www.informatica.si/index.php/informatica/article/download/1875/1105",0,"",""],["Superintelligence As a Cause or Cure For Risks of Astronomical Suffering","Kaj Sotala and Lukas Gloor","2017","report","informatica.si","www.informatica.si/index.php/informatica/article/download/1877/1098",0,"",""],["2017 AI Safety Literature Review and Charity Comparison","Larks","2017","blog","LessWrong","www.lesswrong.com/posts/hYekqQ9hLmn3XTZrp/2017-ai-safety-literature-review-and-charity-comparison",0,"",""],["Human-level hardware timeline","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/human-level-hardware-timeline/",0,"","forecasting"],["Towards an unanimous international regulatory body for responsible use of Artificial Intelligence [UIRB-AI]","Rajesh Chidambaram","2017","paper","arXiv preprint","arxiv.org/abs/1712.07752",0,"","policy"],["Indifference' methods for managing agent rewards","Stuart Armstrong and Xavier O'Rourke","2017","paper","arXiv preprint","arxiv.org/abs/1712.06365",0,"","agents"],["Pascal’s Muggle Pays","Zvi","2017","blog","LessWrong","www.lesswrong.com/posts/CaPgNwxEFHh3Ahvf7/pascal-s-muggle-pays",0,"","theory"],["A Berkeley View of Systems Challenges for AI","Ion Stoica and 13 others","2017","paper","arXiv preprint","arxiv.org/abs/1712.05855",0,"",""],["End-of-the-year matching challenge!","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/12/14/end-of-the-year-matching/",0,"",""],["Occam's razor is insufficient to infer the preferences of irrational agents","Stuart Armstrong and Sören Mindermann","2017","paper","arXiv preprint","arxiv.org/abs/1712.05812",0,"","agents policy"],["Targeted Backdoor Attacks on Deep Learning Systems Using Data Poisoning","","2017","paper","arXiv preprint","arxiv.org/abs/1712.05526",0,"","evals training-data"],["Against the Linear Utility Hypothesis and the Leverage Penalty","AlexMennen","2017","blog","LessWrong","www.lesswrong.com/posts/8FRzErffqEW9gDCCW/against-the-linear-utility-hypothesis-and-the-leverage",0,"","theory"],["Three IQs of AI Systems and their Testing Methods","Feng Liu and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1712.06440",0,"","evals"],["Guarding Slack vs Substance","Raemon","2017","blog","LessWrong","www.lesswrong.com/posts/MeWtcyX8wHxjpDAeE/guarding-slack-vs-substance",0,"","goodharts-law"],["Oracle paper","Stuart_Armstrong","2017","blog","LessWrong","www.lesswrong.com/posts/RcbpeYJMdvxCpTocg/oracle-paper",0,"",""],["A Low-Cost Ethics Shaping Approach for Designing Reinforcement Learning Agents","Yueh-Hua Wu and Shou-De Lin","2017","paper","arXiv preprint","arxiv.org/abs/1712.04172",0,"","agents policy"],["Chance date bias","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/chance-date-bias/",0,"",""],["ML Living Library Opening","Alex Vermeer","2017","blog","intelligence.org","intelligence.org/2017/12/12/ml-living-library/",0,"",""],["Safety models and accident models","Eric Marsden","2017","report","risk-engineering.org","risk-engineering.org/safety-models/",0,"",""],["AI Safety and Reproducibility: Establishing Robust Foundations for the Neuropsychology of Human Values","Gopal P. Sarma and 2 others","2017","paper","In: Gallina B., Skavhaug A., Schoitsch E., Bitsch F. (eds)\n  Computer Safety, Reliability, and Security. SAFECOMP 2018. Lecture Notes in\n  Computer Science, vol 11094. Springer, Cham","arxiv.org/abs/1712.04307",0,"",""],["A reply to Francois Chollet on intelligence explosion","Eliezer Yudkowsky","2017","blog","intelligence.org","intelligence.org/2017/12/06/chollet/",0,"",""],["December 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/12/06/december-2017-newsletter/",0,"",""],["Using Artificial Intelligence to Augment Human Intelligence","Shan Carter and Michael Nielsen","2017","report","Distill","distill.pub/2017/aia",0,"",""],["Implementation of Moral Uncertainty in Intelligent Machines","Kyle Bogosian","2017","report","doi.org","doi.org/10.1007/s11023-017-9448-z",0,"",""],["MIRI’s 2017 Fundraiser","Malo Bourgon","2017","blog","intelligence.org","intelligence.org/2017/12/01/miris-2017-fundraiser/",0,"",""],["Policy Selection Solves Most Problems","abramdemski","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037550c/policy-selection-solves-most-problems",0,"","agents policy theory"],["GoCAS talk on AI Impacts findings","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/gocas-talk-on-ai-impacts-findings/",0,"",""],["AI Safety Gridworlds","Jan Leike and 7 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.09883",0,"","evals agents robustness"],["Price performance Moore’s Law seems slow","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/price-performance-moores-law-seems-slow/",0,"",""],["Sequence Modeling with CTC","Awni Hannun","2017","report","Distill","distill.pub/2017/ctc",0,"",""],["Security Mindset and the Logistic Success Curve","Eliezer Yudkowsky","2017","blog","intelligence.org","intelligence.org/2017/11/26/security-mindset-and-the-logistic-success-curve/",0,"",""],["Security Mindset and Ordinary Paranoia","Eliezer Yudkowsky","2017","blog","intelligence.org","intelligence.org/2017/11/25/security-mindset-ordinary-paranoia/",0,"",""],["The Darwin Results","Zvi","2017","blog","LessWrong","www.lesswrong.com/posts/LDngQb2AJsjTZnWEP/the-darwin-results",0,"","theory"],["Timeless Modesty?","abramdemski","2017","blog","LessWrong","www.lesswrong.com/posts/4vv95qgg9pGWo4eBk/timeless-modesty",0,"","theory"],["Building Machines that Learn and Think for Themselves: Commentary on Lake et al., Behavioral and Brain Sciences, 2017","M. Botvinick and 18 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.08378",0,"","agents"],["Deterministic Policy Optimization by Combining Pathwise and Score Function Estimators for Discrete Action Spaces","Daniel Levy and Stefano Ermon","2017","paper","arXiv preprint","arxiv.org/abs/1711.08068",0,"","policy"],["Evaluating Robustness of Neural Networks with Mixed Integer Programming","Vincent Tjeng and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.07356",0,"","evals robustness"],["Leave no Trace: Learning to Reset for Safe and Autonomous Reinforcement Learning","Benjamin Eysenbach and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.06782",0,"","agents policy"],["Announcing “Inadequate Equilibria”","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/11/16/announcing-inadequate-equilibria/",0,"",""],["The Happy Dance Problem","abramdemski","2017","blog","LessWrong","www.lesswrong.com/posts/HY94LBqekihnx85WQ/the-happy-dance-problem",0,"","theory"],["Using KL-divergence to focus Deep Visual Explanation","Housam Khalifa Bashier Babiker and Randy Goebel","2017","paper","arXiv preprint","arxiv.org/abs/1711.06431",0,"","interpretability evals"],["Good and safe uses of AI Oracles","Stuart Armstrong and Xavier O'Rorke","2017","paper","arXiv preprint","arxiv.org/abs/1711.05541",0,"","agents robustness"],["Fixing Weight Decay Regularization in Adam","Ilya Loshchilov & Frank Hutter","2017","paper","arXiv preprint","arxiv.org/abs/1711.05101",0,"",""],["Rationalising humans: another mugging, but not Pascal's","Stuart_Armstrong","2017","blog","LessWrong","www.lesswrong.com/posts/inr5wznBNNipSyYEM/rationalising-humans-another-mugging-but-not-pascal-s",0,"",""],["Military AI as a Convergent Goal of Self-Improving AI","avturchin","2017","blog","LessWrong","www.lesswrong.com/posts/YZ28xp6XDiD9fNwpn/military-ai-as-a-convergent-goal-of-self-improving-ai",0,"","instrumental-convergence"],["2017 trend in the cost of computing","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/recent-trend-in-the-cost-of-computing/",0,"",""],["A Survey of Artificial General Intelligence Projects for Ethics, Risk, and Policy","Seth Baum","2017","report","papers.ssrn.com","papers.ssrn.com/abstract=3070741",0,"","policy"],["A major grant from the Open Philanthropy Project","Malo Bourgon","2017","blog","intelligence.org","intelligence.org/2017/11/08/major-grant-open-phil/",0,"",""],["Price-performance trend in top supercomputers","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/price-performance-trend-in-top-supercomputers/",0,"",""],["Feature Visualization","Chris Olah and 2 others","2017","report","Distill","distill.pub/2017/feature-visualization",0,"","mechanistic-interpretability"],["November 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/11/03/november-2017-newsletter/",0,"",""],["On the promotion of safe and socially beneficial artificial intelligence","Seth D. Baum","2017","report","link.springer.com","link.springer.com/10.1007/s00146-016-0677-0",0,"",""],["A Foundry of Human Activities and Infrastructures","Robert B. Allen and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.01927",0,"",""],["Mixed-Strategy Ratifiability Implies CDT=EDT","abramdemski","2017","blog","LessWrong","www.lesswrong.com/posts/x2wn2MWYSafDtm8Lf/mixed-strategy-ratifiability-implies-cdt-edt-1",0,"","theory"],["Servant of Many Masters: Shifting priorities in Pareto-optimal sequential decision-making","Andrew Critch and Stuart Russell","2017","paper","arXiv preprint","arxiv.org/abs/1711.00363",0,"","agents policy"],["Learning Robust Rewards with Adversarial Inverse Reinforcement Learning","Justin Fu and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1710.11248",0,"","agents policy"],["Tokyo AI & Society Symposium","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/10/30/tokyo-ai-society-symposium/",0,"",""],["Logical Updatelessness as a Robust Delegation Problem","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/K5Qp7ioupgb7r73Ca/logical-updatelessness-as-a-robust-delegation-problem",0,"","theory"],["Computing hardware performance data collections","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/computing-hardware-performance-data-collections/",0,"",""],["Zero-Knowledge Cooperation","bryjnar","2017","blog","LessWrong","www.lesswrong.com/posts/TDHDWMP5PRk4f3zrR/zero-knowledge-cooperation",0,"","theory"],["Humans can be assigned any values whatsoever...","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754e8/humans-can-be-assigned-any-values-whatsoever",0,"",""],["Human-in-the-loop Artificial Intelligence","Fabio Massimo Zanzotto","2017","paper","Journal of Artificial Intelligence Research, 2019","arxiv.org/abs/1710.08191",0,"",""],["A behaviorist approach to building phenomenological bridges","Caspar Oesterheld","2017","report","casparoesterheld.com","casparoesterheld.com/2017/10/22/a-behaviorist-approach-to-building-phenomenological-bridges/",0,"",""],["New paper: “Functional Decision Theory”","Matthew Graves","2017","blog","intelligence.org","intelligence.org/2017/10/22/fdt/",0,"","theory"],["What Evidence Is AlphaGo Zero Re AGI Complexity?","RobinHanson","2017","blog","LessWrong","www.lesswrong.com/posts/D3NspiH2nhKA6B2PE/what-evidence-is-alphago-zero-re-agi-complexity",0,"","forecasting"],["AlphaGo Zero and the Foom Debate","Eliezer Yudkowsky","2017","blog","intelligence.org","intelligence.org/2017/10/20/alphago/",0,"",""],["AlphaGo Zero and the Foom Debate","Eliezer Yudkowsky","2017","blog","LessWrong","www.lesswrong.com/posts/shnSyzv4Jq3bhMNw5/alphago-zero-and-the-foom-debate",0,"","forecasting"],["AlphaGo Zero and capability amplification","Paul Christiano","2017","report","ai-alignment.com","ai-alignment.com/alphago-zero-and-capability-amplification-ede767bb8446",0,"","scalable-oversight"],["Functional Decision Theory: A New Theory of Instrumental Rationality","ESRogs","2017","blog","LessWrong","www.lesswrong.com/posts/AGAGgoWymRhJ5Rqyv/functional-decision-theory-a-new-theory-of-instrumental",0,"","theory"],["Decision Trees for Helpdesk Advisor Graphs","Spyros Gkezerlis and Dimitris Kalles","2017","paper","Bulletin of the Technical Committee on Learning Technology, Volume\n  18, Issue 2-3, April 2016","arxiv.org/abs/1710.07075",0,"","agents"],["Yudkowsky on AGI ethics","Rob Bensinger","2017","blog","LessWrong","www.lesswrong.com/posts/SsCQHjqNT3xQAPQ6b/yudkowsky-on-agi-ethics",0,"","forecasting"],["October 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/10/16/october-2017-newsletter/",0,"",""],["Why no total winner?","Paul Crowley","2017","blog","LessWrong","www.lesswrong.com/posts/As76yueYGy6FjZg3R/why-no-total-winner",0,"","forecasting"],["There's No Fire Alarm for Artificial General Intelligence","EA Forum Archives","2017","blog","EA Forum","forum.effectivealtruism.org/posts/cPuXn8oDJpTDxGGdB/there-s-no-fire-alarm-for-artificial-general-intelligence",0,"",""],["There’s No Fire Alarm for Artificial General Intelligence","Eliezer Yudkowsky","2017","blog","intelligence.org","intelligence.org/2017/10/13/fire-alarm/",0,"",""],["Functional Decision Theory: A New Theory of Instrumental Rationality","Eliezer Yudkowsky and Nate Soares","2017","paper","arXiv preprint","arxiv.org/abs/1710.05060",0,"","theory"],["Robot Sex: Social and Ethical Implications","John Danaher and Neil McArthur","2017","report","goodreads.com","www.goodreads.com/book/show/34540069-robot-sex",0,"",""],["There's No Fire Alarm for Artificial General Intelligence","Eliezer Yudkowsky","2017","blog","LessWrong","www.lesswrong.com/posts/BEtzRE2M5m9YEAQpX/there-s-no-fire-alarm-for-artificial-general-intelligence",0,"","forecasting"],["Consequence assessment: Estimating the impact of accident scenarios","Eric Marsden","2017","report","risk-engineering.org","risk-engineering.org/consequence-assessment/",0,"",""],["Toy model of the AI control problem: animated version","Stuart_Armstrong","2017","blog","LessWrong","www.lesswrong.com/posts/EdEhGPEJi6dueQXv2/toy-model-of-the-ai-control-problem-animated-version",0,"","ai-control deception"],["Delegative Reinforcement Learning with a Merely Sane Advisor","Vanessa Kosoy","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754d5/delegative-reinforcement-learning-with-a-merely-sane-advisor",0,"",""],["2016 ESPAI Narrow AI task forecast timeline","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/2016-espai-narrow-ai-task-forecast-timeline/",0,"","forecasting"],["Social Choice Ethics in Artificial Intelligence (paper challenging CEV-like approaches to choosing an AI's values)","Kaj_Sotala","2017","blog","LessWrong","www.lesswrong.com/posts/8vNtxXRrH4v6rP96z/social-choice-ethics-in-artificial-intelligence-paper",0,"",""],["Global Catastrophes: The Most Extreme Risks","Seth Baum and Anthony Barrett","2017","report","papers.ssrn.com","papers.ssrn.com/abstract=3046668",0,"",""],["An intervention to shape policy dialogue, communication, and AI research norms for AI safety","Lee_Sharkey","2017","blog","EA Forum","forum.effectivealtruism.org/posts/4kRPYuogoSKnHNBhY/an-intervention-to-shape-policy-dialogue-communication-and",0,"","governance policy"],["CHAI Newsletter 2017","CHAI","2017","report","drive.google.com","drive.google.com/file/d/1GSpRS-No3ODE2XRQBkYCDf7KZ9zSOnbb/view?usp=sharing",0,"",""],["How feasible is the rapid development of artificial superintelligence?","Kaj Sotala","2017","report","doi.org","doi.org/10.1088%2F1402-4896%2Faa90e8",0,"",""],["Deep TAMER: Interactive Agent Shaping in High-Dimensional State Spaces","Garrett Warnell and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1709.10163",0,"","agents training-data"],["What do ML researchers think you are wrong about?","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/what-do-ml-researchers-think-you-are-wrong-about/",0,"",""],["When do ML Researchers Think Specific Tasks will be Automated?","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/when-do-ml-researchers-think-specific-tasks-will-be-automated/",0,"",""],["Analogpunk","Tamsin Leake","2017","blog","carado.moe","carado.moe/analogpunk.html",0,"",""],["September 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/09/24/september-2017-newsletter/",0,"",""],["Autonomous Agents Modelling Other Agents: A Comprehensive Survey and Open Problems","Stefano V. Albrecht and Peter Stone","2017","paper","arXiv preprint","arxiv.org/abs/1709.08071",0,"","deception agents"],["Naturalized induction – a challenge for evidential and causal decision theory","Caspar Oesterheld","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/kgsaSbJqWLtJfiCcz/naturalized-induction-a-challenge-for-evidential-and-causal",0,"","theory"],["A Voting-Based System for Ethical Decision Making","Ritesh Noothigattu and 6 others","2017","paper","arXiv preprint","arxiv.org/abs/1709.06692",0,"","evals"],["Incorrigibility in the CIRL Framework","Ryan Carey","2017","paper","arXiv preprint","arxiv.org/abs/1709.06275",0,"",""],["DropoutDAgger: A Bayesian Approach to Safe Imitation Learning","Kunal Menda and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1709.06166",0,"","evals policy training-data"],["A Learning and Masking Approach to Secure Learning","Linh Nguyen and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1709.04447",0,"","robustness"],["Automation of music production","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/automation-of-music-production/",0,"",""],["Aggregating incoherent agents who disagree","Richard Pettigrew","2017","paper","arXiv preprint","arxiv.org/abs/1709.03981",0,"","agents"],["Stuart Russell’s description of AI risk","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/stuart-russells-description-of-ai-risk/",0,"",""],["Knowledge Transfer Between Artificial Intelligence Systems","Ivan Y. Tyukin and 3 others","2017","paper","Front Neurorobot. 2018; 12: 49","arxiv.org/abs/1709.01547",0,"","robustness"],["New paper: “Incorrigibility in the CIRL Framework”","Matthew Graves","2017","blog","intelligence.org","intelligence.org/2017/08/31/incorrigibility-in-cirl/",0,"",""],["Value of Global Catastrophic Risk (GCR) Information: Cost-Effectiveness-Based Approach for GCR Reduction","Anthony Michael Barrett","2017","report","pubsonline.informs.org","pubsonline.informs.org/doi/10.1287/deca.2017.0350",0,"",""],["Why Does Deep and Cheap Learning Work So Well?","Henry W. Lin and 2 others","2017","report","link.springer.com","link.springer.com/10.1007/s10955-017-1836-5",0,"",""],["The Doomsday argument in anthropic decision theory","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754d4/the-doomsday-argument-in-anthropic-decision-theory",0,"","theory"],["Artificial intelligence: The future is superintelligent [Book review of \"Life 3.0: Being Human in the Age of Artificial Intelligence\" by Max Tegmark]","Stuart Russell","2017","report","nature.com","www.nature.com/articles/548520a",0,"",""],["Safe Reinforcement Learning via Shielding","Mohammed Alshiekh and 5 others","2017","paper","arXiv preprint","arxiv.org/abs/1708.08611",0,"","agents"],["Using modal fixed points to formalize logical causality","cousin_it","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374e61/using-modal-fixed-points-to-formalize-logical-causality",0,"",""],["Life 3.0: Being Human in the Age of Artificial Intelligence (2017, Alfred A. Knopf)","Max Teqmark","2017","report","goodreads.com","www.goodreads.com/book/show/34272565-life-3-0",0,"",""],["BadNets: Identifying Vulnerabilities in the Machine Learning Model Supply Chain","Tianyu Gu and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1708.06733",0,"",""],["Logical Induction with incomputable sequences","AlexMennen","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754b7/logical-induction-with-incomputable-sequences",0,"","theory"],["On Ensuring that Intelligent Machines Are Well-Behaved","Philip S. Thomas and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1708.05448",0,"",""],["Stable Pointers to Value: An Agent Embedded in Its Own Utility Function","abramdemski","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703754b3/stable-pointers-to-value-an-agent-embedded-in-its-own-utility-function",0,"","agents"],["August 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/08/16/august-2017-newsletter/",0,"",""],["Portfolio approach to AI safety research","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/08/16/portfolio-approach-to-ai-safety-research/",0,"",""],["Inverse Reward Design.","Dylan Hadfield-Menell and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1711.02827",0,"","agents"],["Actual Causality (Book).","Joseph Y and Halpern","2017","report","mitpress.mit.edu","mitpress.mit.edu/books/actual-causality",0,"",""],["Causality, Responsibility and Blame in Team Plans.","Natasha Alechina and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/2005.10297",0,"","agents"],["Comparing Human-Centric and Robot-Centric Sampling for Robot Deep Learning from Demonstrations.","Michael Laskey and 7 others","2017","paper","arXiv preprint","arxiv.org/abs/1610.00850",0,"","policy"],["Computational Extensive-Form Games.","Joseph Y and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1506.03030",0,"","deception"],["Do You Want Your Autonomous Car to Drive Like You?.","Chandrayee Basu and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1802.01636",0,"",""],["Expressive Robot Motion Timing.","Allan Zhou and 3 others","2017","paper","HRI '17 Proceedings of the 2017 ACM/IEEE International Conference\n  on Human-Robot Interaction Pages 22-31","arxiv.org/abs/1802.01536",0,"",""],["Modeling Agents with Probabilistic Programs.","Owain Evans and 3 others","2017","report","agentmodels.org","agentmodels.org/",0,"","agents"],["Repeated Inverse Reinforcement Learning.","Kareem Amin and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.05427",0,"","agents"],["Self-confirming price-prediction strategies for simultaneous one-shot auctions.","Michael Wellman and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1210.4915",0,"","agents robustness"],["The Computational Complexity of Structure-Based Causality.","Gadi Aleksandrowicz and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1412.3076",0,"",""],["Toward a Rational and Mechanistic Account of Mental Effort.","Amitai Shenhav and 6 others","2017","report","pubmed.ncbi.nlm.nih.gov","pubmed.ncbi.nlm.nih.gov/28375769/",0,"",""],["Translating Neuralese.","Jacob Andreas and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1704.06960",0,"","agents"],["Potential Risks from Advanced AI","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/iDYt2e4skogJEn946/potential-risks-from-advanced-ai",0,"","governance"],["What does (and doesn't) AI mean for effective altruism?","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/Mw9ZxmZqiaXM2rb49/what-does-and-doesn-t-ai-mean-for-effective-altruism",0,"","governance"],["Daniel Dewey: The Open Philanthropy Project's work on potential risks from advanced AI","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/fqEcHtEvancXg4Jy4/daniel-dewey-the-open-philanthropy-project-s-work-on",0,"","governance"],["Jan Leike, Helen Toner, Malo Bourgon, and Miles Brundage: Working in AI","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/izD5LT6qvyfTqyCKv/jan-leike-helen-toner-malo-bourgon-and-miles-brundage",0,"",""],["Katja Grace: AI safety","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/KC5PhJANXhiwbhMq5/katja-grace-ai-safety",0,"",""],["Michael Page, Dario Amodei, Helen Toner, Tasha McCauley, Jan Leike, & Owen Cotton-Barratt: Musings on AI","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/qZN4opfZs7iZfJkY6/michael-page-dario-amodei-helen-toner-tasha-mccauley-jan",0,"",""],["Owen Cotton-Barratt: What does (and doesn't) AI mean for effective altruism?","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/DGQHZZNMdjDghgu2S/owen-cotton-barratt-what-does-and-doesn-t-ai-mean-for",0,"","governance"],["Active Preference-Based Learning of Reward Functions.","Dorsa Sadigh and 4 others","2017","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~sastry/pubs/Pdfs%20of%202017/SadighActive2017.pdf",0,"",""],["An automatic method for discovering rational heuristics for risky choice.","Falk Lieder and 2 others","2017","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/Meta_Decision_Making-CameraReady.pdf",0,"",""],["Enhancing metacognitive reinforcement learning using reward structures and feedback.","Paul Krueger and 2 others","2017","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/Accelerating_Metacognitive_RL-CameraReady.pdf",0,"",""],["Multiverse-wide Cooperation via Correlated Decision Making","Caspar Oesterheld","2017","report","longtermrisk.org","longtermrisk.org/files/Multiverse-wide-Cooperation-via-Correlated-Decision-Making.pdf",0,"",""],["Reasoning about Rationality.","Adam Bjorndahl and 3 others","2017","report","cs.cornell.edu","www.cs.cornell.edu/Info/People/halpern/papers/axrat.pdf",0,"",""],["The evolution of cognitive mechanisms in response to cultural innovations.","Arnon Lotem and 4 others","2017","report","doi.org","doi.org/10.1073/pnas.1620742114",0,"",""],["The Structure of Goal Systems Predicts Human Performance.","David Bourgin and 4 others","2017","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/cogsciReichman.pdf",0,"",""],["When Does Bounded-Optimal Metareasoning Favor Few Cognitive Systems?.","Smitha Milli and 2 others","2017","report","cocosci.princeton.edu","cocosci.princeton.edu/papers/Milli_AAAI.pdf",0,"",""],["CIRL Wireheading","tom4everitt","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375410/cirl-wireheading",0,"","reward-hacking"],["Designing for safety","Eric Marsden","2017","report","risk-engineering.org","risk-engineering.org/safe-design/",0,"",""],["A Formal Approach to the Problem of Logical Non-Omniscience","Scott Garrabrant and 4 others","2017","paper","EPTCS 251, 2017, pp. 221-235","arxiv.org/abs/1707.08747",0,"","deception theory"],["Together We Know How to Achieve: An Epistemic Logic of Know-How (Extended Abstract)","Pavel Naumov and Jia Tao","2017","paper","EPTCS 251, 2017, pp. 441-453","arxiv.org/abs/1707.08759",0,"",""],["The future of growth: near-zero growth rates","Center on Long-Term Risk","2017","report","longtermrisk.org","longtermrisk.org/the-future-of-growth-near-zero-growth-rates/",0,"",""],["Using Program Induction to Interpret Transition System Dynamics","Svetlin Penkov and Subramanian Ramamoorthy","2017","paper","arXiv preprint","arxiv.org/abs/1708.00376",0,"","interpretability agents"],["July 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/07/25/july-2017-newsletter/",0,"",""],["Guidelines for Artificial Intelligence Containment","James Babcock and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1707.08476",0,"","agents"],["Adversarial Examples for Evaluating Reading Comprehension Systems","Robin Jia","2017","paper","arXiv preprint","arxiv.org/abs/1707.07328",0,"","evals robustness"],["Pragmatic-Pedagogic Value Alignment","Jaime F. Fisac and 9 others","2017","paper","International Symposium on Robotics Research, 2017","arxiv.org/abs/1707.06354",0,"","agents theory"],["RAIL: Risk-Averse Imitation Learning","Anirban Santara and 6 others","2017","paper","arXiv preprint","arxiv.org/abs/1707.06658",0,"","evals benchmarks agents"],["Logic Programming approaches for routing fault-free and maximally-parallel Wavelength Routed Optical Networks on Chip (Application paper)","Marco Gavanelli and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1707.05858",0,"","deception robustness"],["Open Problems Regarding Counterfactuals: An Introduction For Beginners","Diffractor","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375494/open-problems-regarding-counterfactuals-an-introduction-for-beginners",0,"",""],["Trial without Error: Towards Safe Reinforcement Learning via Human Intervention","William Saunders and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1707.05173",0,"","evals agents robustness"],["Delegative Inverse Reinforcement Learning","Vanessa Kosoy","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037546b/delegative-inverse-reinforcement-learning",0,"",""],["My current thoughts on MIRI's \"highly reliable agent design\" work","Daniel_Dewey","2017","blog","EA Forum","forum.effectivealtruism.org/posts/SEL9PW8jozrvLnkb4/my-current-thoughts-on-miri-s-highly-reliable-agent-design",0,"","agents"],["Updates to the research team, and a major donation","Malo Bourgon","2017","blog","intelligence.org","intelligence.org/2017/07/04/updates-to-the-research-team-and-a-major-donation/",0,"",""],["Approval-maximizing representations","Paul Christiano","2017","report","ai-alignment.com","ai-alignment.com/approval-maximizing-representations-56ee6a6a1fe6",0,"",""],["Artificial Intelligence and Global Security Initiative Research Agenda","Center for a New American Security","2017","report","cnas.org","www.cnas.org/artificial-intelligence-and-global-security-initiative-research-agenda",0,"",""],["Teacher-Student Curriculum Learning","Tambet Matiisen and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1707.00183",0,"",""],["AI hopes and fears in numbers","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/ai-hopes-and-fears-in-numbers/",0,"",""],["2016 ESPAI questions printout","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/2016-esopai-questions-printout/",0,"",""],["A survey of polls on Newcomb’s problem","Caspar","2017","report","casparoesterheld.com","casparoesterheld.com/2017/06/27/a-survey-of-polls-on-newcombs-problem/",0,"",""],["Complications in evaluating neglectedness","Caspar Oesterheld","2017","report","casparoesterheld.com","casparoesterheld.com/2017/06/25/complications-in-evaluating-neglectedness/",0,"","evals"],["Expert and Non-Expert Opinion about Technological Unemployment","Toby Walsh","2017","paper","arXiv preprint","arxiv.org/abs/1706.06906",0,"",""],["Towards Deep Learning Models Resistant to Adversarial Attacks","Aleksander Madry and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1706.06083",0,"","robustness"],["The AI revolution and international politics _ Allan Dafoe _ EAG 2017 Boston-by Centre for Effective Altruism-video_id Zef-mIKjHAk-date 20170618","Allan Dafoe","2017","report","drive.google.com","drive.google.com/file/d/1UqRKpGkRBGtqqlUAYf1OUbkpTcfRK7se/view?usp=share_link",0,"",""],["June 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/06/16/june-2017-newsletter/",0,"",""],["Media discussion of 2016 ESPAI","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/media-discussion-of-2016-espai/",0,"",""],["Device Placement Optimization with Reinforcement Learning","Azalia Mirhoseini and 9 others","2017","paper","arXiv preprint","arxiv.org/abs/1706.04972",0,"",""],["Attention Is All You Need","Ashish Vaswani and 7 others","2017","paper","arXiv preprint","arxiv.org/abs/1706.03762",0,"","deception training-data"],["Responsible Autonomy","Virginia Dignum","2017","paper","arXiv preprint","arxiv.org/abs/1706.02513",0,"","agents"],["Some survey results!","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/some-survey-results/",0,"",""],["SSC Journal Club: AI Timelines","Scott Alexander","2017","blog","LessWrong","www.lesswrong.com/posts/qL8Z9TBCNWQyN6yLq/ssc-journal-club-ai-timelines",0,"","forecasting"],["Cognitive Science/Psychology As a Neglected Approach to AI Safety","Kaj_Sotala","2017","blog","EA Forum","forum.effectivealtruism.org/posts/WdMnmmqqiP5zCtSfv/cognitive-science-psychology-as-a-neglected-approach-to-ai",0,"",""],["Takeaways from self-tracking data","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/06/04/takeaways-from-self-tracking-data/",0,"",""],["Cooperative Oracles: Introduction","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375419/cooperative-oracles-introduction",0,"",""],["Cooperative Oracles: Nonexploited Bargaining","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037541a/cooperative-oracles-nonexploited-bargaining",0,"",""],["Cooperative Oracles: Stratified Pareto Optima and Almost Stratified Pareto Optima","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375441/cooperative-oracles-stratified-pareto-optima-and-almost-stratified-pareto-optima",0,"",""],["Acausal trade: different utilities, different trades","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375415/acausal-trade-different-utilities-different-trades",0,"",""],["Acausal trade: double decrease","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375414/acausal-trade-double-decrease",0,"",""],["Acausal trade: universal utility, or selling non-existence insurance too late","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037541c/acausal-trade-universal-utility-or-selling-non-existence-insurance-too-late",0,"",""],["Benign model-free RL","Paul Christiano","2017","report","ai-alignment.com","ai-alignment.com/benign-model-free-rl-4aae8c97e385",0,"",""],["Corrigibility thoughts I: caring about multiple things","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037531d/corrigibility-thoughts-i-caring-about-multiple-things",0,"",""],["Counterfactually uninfluenceable agents","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037536b/counterfactually-uninfluenceable-agents",0,"","agents"],["Counterfactuals on POMDP","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703752a9/counterfactuals-on-pomdp",0,"",""],["The AI revolution and international politics (Allan Dafoe)","EA Global","2017","blog","EA Forum","forum.effectivealtruism.org/posts/3538iKtS2YmN67som/the-ai-revolution-and-international-politics-allan-dafoe",0,"","governance policy"],["Thoughts on Quantilizers","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037532c/thoughts-on-quantilizers",0,"",""],["The Social Science of Computerized Brains – Review of The Age of Em: Work, Love, and Life When Robots Rule the Earth by Robin Hanson (Oxford University Press, 2016)","Seth D. Baum","2017","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0016328716302518",0,"",""],["Why I am not currently working on the AAMLS agenda","jessicata","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037541b/why-i-am-not-currently-working-on-the-aamls-agenda",0,"",""],["The Atari Grand Challenge Dataset","Vitaly Kurin and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.10998",0,"","evals robustness"],["The Singularity May Be Near","Roman V. Yampolskiy","2017","paper","arXiv preprint","arxiv.org/abs/1706.01303",0,"",""],["Constrained Policy Optimization","Joshua Achiam and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.10528",0,"","agents policy"],["Futarchy Fix","abramdemski","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375432/futarchy-fix",0,"",""],["Low Impact Artificial Intelligences","Stuart Armstrong and Benjamin Levinstein","2017","paper","arXiv preprint","arxiv.org/abs/1705.10720",0,"","ai-control"],["Universal Reinforcement Learning Algorithms: Survey and Experiments","John Aslanides and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.10557",0,"","agents"],["The Technological Singularity: Managing the Journey","Vic Callaghan and 3 others","2017","report","goodreads.com","www.goodreads.com/book/show/32850550-the-technological-singularity",0,"",""],["Should Robots be Obedient?","Smitha Milli and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.09990",0,"","deception robustness"],["Existential risk from AI without an intelligence explosion","AlexMennen","2017","blog","LessWrong","www.lesswrong.com/posts/bFcbG2TQCCE3krhEY/existential-risk-from-ai-without-an-intelligence-explosion",0,"",""],["Together We Know How to Achieve: An Epistemic Logic of Know-How","Pavel Naumov and Jia Tao","2017","paper","arXiv preprint","arxiv.org/abs/1705.09349",0,"",""],["Reflexive Oracles and superrationality: Pareto","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375068/reflexive-oracles-and-superrationality-pareto",0,"",""],["Reflexive Oracles and superrationality: prisoner's dilemma","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037505e/reflexive-oracles-and-superrationality-prisoner-s-dilemma",0,"",""],["When Will AI Exceed Human Performance? Evidence from AI Experts","Katja Grace and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.08807",0,"","policy"],["Reinforcement Learning with a Corrupted Reward Channel","Tom Everitt and 4 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.08417",0,"","agents"],["Thinking Fast and Slow with Deep Learning and Tree Search","Thomas Anthony and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.08439",0,"","agents policy"],["Acausal trade: being unusual","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703753d9/acausal-trade-being-unusual",0,"",""],["Acausal trade: conclusion: theory vs practice","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375427/acausal-trade-conclusion-theory-vs-practice",0,"",""],["Acausal trade: full decision algorithms","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375417/acausal-trade-full-decision-algorithms",0,"",""],["Strategically knowing how","Raul Fervari and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.05254",0,"","agents"],["Anthropic uncertainty in the Evidential Blackmail","Johannes Treutlein","2017","report","casparoesterheld.com","casparoesterheld.com/2017/05/12/anthropic-uncertainty-in-the-evidential-blackmail/",0,"",""],["Forecasting using incomplete models","Vanessa Kosoy","2017","paper","arXiv preprint","arxiv.org/abs/1705.04630",0,"","forecasting"],["Acausal trade: Introduction","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375416/acausal-trade-introduction",0,"",""],["Robot Planning with Mathematical Models of Human State and Action","Anca D. Dragan","2017","paper","arXiv preprint","arxiv.org/abs/1705.04226",0,"",""],["May 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/05/10/may-2017-newsletter/",0,"",""],["Infinite ethics comparisons","Stuart_Armstrong","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037540c/infinite-ethics-comparisons",0,"",""],["Informatica: Special Issue on Superintelligence","RyanCarey","2017","blog","EA Forum","forum.effectivealtruism.org/posts/vDsGvWEzoccPnJqDQ/informatica-special-issue-on-superintelligence",0,"",""],["Finding reflective oracle distributions using a Kakutani map","jessicata","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375401/finding-reflective-oracle-distributions-using-a-kakutani-map",0,"",""],["2017 Updates and Strategy","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/04/30/2017-updates-and-strategy/",0,"",""],["Highlights from the ICLR conference: food, ships, and ML security","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/04/30/highlights-from-the-iclr-conference-food-ships-and-ml-security/",0,"",""],["Software Engineer Internship / Staff Openings","Alex Vermeer","2017","blog","intelligence.org","intelligence.org/2017/04/30/software-engineer-internship-staff-openings/",0,"",""],["That is not dead which can eternal lie: the aestivation hypothesis for resolving Fermi's paradox","Anders Sandberg and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1705.03394",0,"",""],["Network Dissection: Quantifying Interpretability of Deep Visual Representations","David Bau","2017","paper","arXiv preprint","arxiv.org/abs/1704.05796",0,"","interpretability evals"],["Two Major Obstacles for Logical Inductor Decision Theory","Scott Garrabrant","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703753d4/two-major-obstacles-for-logical-inductor-decision-theory",0,"","theory"],["Intro to caring about AI alignment as an EA cause","So8res","2017","blog","EA Forum","forum.effectivealtruism.org/posts/pfEpu3gMG5bRMyfee/intro-to-caring-about-ai-alignment-as-an-ea-cause",0,"",""],["Ensuring smarter-than-human intelligence has a positive outcome","Nate Soares","2017","blog","intelligence.org","intelligence.org/2017/04/12/ensuring/",0,"",""],["Interpretable Explanations of Black Boxes by Meaningful Perturbation","Ruth C. Fong and Andrea Vedaldi","2017","paper","Proceedings of the 2017 IEEE International Conference on Computer\n  Vision (ICCV)","arxiv.org/abs/1704.03296",0,"","interpretability"],["Dynamic Safe Interruptibility for Decentralized Multi-Agent Reinforcement Learning","El Mahdi El Mhamdi and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1704.02882",0,"","agents"],["Decisions are for making bad outcomes inconsistent","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/04/07/decisions-are-for-making-bad-outcomes-inconsistent/",0,"",""],["Guide to pages on AI timeline predictions","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/guide-to-pages-on-ai-timeline-predictions/",0,"","forecasting"],["April 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/04/06/april-2017-newsletter/",0,"",""],["Why Momentum Really Works","Distill","2017","report","Distill","distill.pub/2017/momentum",0,"",""],["G.K. Chesterton On AI Risk","Scott Alexander","2017","report","slatestarcodex.com","slatestarcodex.com/2017/04/01/g-k-chesterton-on-ai-risk/",0,"",""],["Two new researchers join MIRI","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/03/31/two-new-researchers-join-miri/",0,"",""],["On the Impossibility of Supersized Machines","Ben Garfinkel and 8 others","2017","paper","arXiv preprint","arxiv.org/abs/1703.10987",0,"",""],["2016 in review","Malo Bourgon","2017","blog","intelligence.org","intelligence.org/2017/03/28/2016-in-review/",0,"",""],["On Automating the Doctrine of Double Effect","Naveen Sundar Govindarajulu and Selmer Bringsjord","2017","paper","arXiv preprint","arxiv.org/abs/1703.08922",0,"","deception agents"],["Research Debt","Chris Olah and Shan Carter","2017","report","Distill","distill.pub/2017/research-debt",0,"",""],["2016 International Symposium on Experimental Robotics","Dana Kulić and 2 others","2017","report","goodreads.com","www.goodreads.com/book/show/32763109-2016-international-symposium-on-experimental-robotics",0,"",""],["Counterfactual Fairness","Matt J. Kusner and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1703.06856",0,"",""],["New paper: “Cheating Death in Damascus”","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/03/18/new-paper-cheating-death-in-damascus/",0,"",""],["Sets with Small Intersection","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/03/17/sets-with-small-intersection/",0,"",""],["March 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/03/15/march-2017-newsletter/",0,"",""],["Progress in general purpose factoring","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/progress-in-general-purpose-factoring/",0,"",""],["The average utilitarian’s solipsism wager","Caspar","2017","report","casparoesterheld.com","casparoesterheld.com/2017/03/15/the-average-utilitarians-solipsism-wager/",0,"",""],["HCH as a measure of manipulation","orthonormal","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375393/hch-as-a-measure-of-manipulation",0,"",""],["Right for the Right Reasons: Training Differentiable Models by Constraining their Explanations","Andrew Slavin Ross and 2 others","2017","paper","arXiv preprint","arxiv.org/abs/1703.03717",0,"",""],["A proposal for ethically traceable artificial intelligence","Christopher A. Tucker","2017","paper","arXiv preprint","arxiv.org/abs/1703.01908",0,"",""],["A model of pathways to artificial superintelligence catastrophe for risk and decision analysis","Anthony M. Barrett and Seth D. Baum","2017","report","tandfonline.com","www.tandfonline.com/doi/full/10.1080/0952813X.2016.1186228",0,"",""],["Generalizing Foundations of Decision Theory","abramdemski","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375373/generalizing-foundations-of-decision-theory",0,"","theory"],["Trends in algorithmic progress","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/trends-in-algorithmic-progress/",0,"",""],["Do You Want Your Autonomous Car to Drive Like You?","C. Basu and 4 others","2017","report","researchgate.net","www.researchgate.net/publication/314159073_Do_You_Want_Your_Autonomous_Car_To_Drive_Like_You",0,"",""],["Using machine learning to address AI risk","Jessica Taylor","2017","blog","intelligence.org","intelligence.org/2017/02/28/using-machine-learning/",0,"",""],["Advice for Authors","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/02/28/advice-for-authors/",0,"",""],["Towards A Rigorous Science of Interpretable Machine Learning","Finale Doshi-Velez and Been Kim","2017","paper","arXiv preprint","arxiv.org/abs/1702.08608",0,"","interpretability evals"],["Don't Fear the Reaper: Refuting Bostrom's Superintelligence Argument","Sebastian Benthall","2017","paper","arXiv preprint","arxiv.org/abs/1702.08495",0,"","agents policy"],["Synergistic Team Composition","Ewa Andrejczuk and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1702.08222",0,"","agents"],["What Should the Average EA Do About AI Alignment?","Raemon","2017","blog","EA Forum","forum.effectivealtruism.org/posts/DkQaJwYMkSFN6E3f9/what-should-the-average-ea-do-about-ai-alignment",0,"",""],["Changes in funding in the AI safety field","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/changes-in-funding-in-the-ai-safety-field/",0,"",""],["Funding of AI Research","Katja Grace","2017","blog","aiimpacts.org","aiimpacts.org/funding-of-ai-research/",0,"",""],["February 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/02/16/february-2017-newsletter/",0,"",""],["CHCAI/MIRI research internship in AI safety","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/02/11/chcai-miri/",0,"",""],["Enabling Robots to Communicate their Objectives","Sandy H. Huang and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1702.03465",0,"","deception"],["Model Mis-specification and Inverse Reinforcement Learning","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/02/07/model-mis-specification-and-inverse-reinforcement-learning/",0,"",""],["Linear algebra fact","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/02/06/linear-algebra-fact/",0,"",""],["“Betting on the Past” by Arif Ahmed","Johannes Treutlein","2017","report","casparoesterheld.com","casparoesterheld.com/2017/02/06/betting-on-the-past-by-arif-ahmed/",0,"",""],["Prékopa–Leindler inequality","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/02/05/prekopa-leindler-inequality/",0,"",""],["Changes in funding in the AI safety field","Sebastian_Farquhar","2017","blog","EA Forum","forum.effectivealtruism.org/posts/Q83ayse5S8CksbT7K/changes-in-funding-in-the-ai-safety-field",0,"",""],["My current take on the Paul-MIRI disagreement on alignability of messy AI","jessicata","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703752c6/my-current-take-on-the-paul-miri-disagreement-on-alignability-of-messy-ai",0,"",""],["On motivations for MIRI's highly reliable agent design research","jessicata","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375321/on-motivations-for-miri-s-highly-reliable-agent-design-research",0,"","agents"],["Plan Explanations as Model Reconciliation: Moving Beyond Explanation as Soliloquy","Tathagata Chakraborti and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1701.08317",0,"","evals"],["Practical Reasoning with Norms for Autonomous Software Agents (Full Edition)","Zohreh Shams and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1701.08306",0,"","agents"],["New paper: “Toward negotiable reinforcement learning”","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/01/25/negotiable-rll/",0,"",""],["Interactive Learning from Policy-Dependent Human Feedback","James MacGlashan and 7 others","2017","paper","International Conference on Machine Learning. PMLR, 2017","arxiv.org/abs/1701.06049",0,"","rlhf policy"],["A measure-theoretic generalization of logical induction","Vanessa Kosoy","2017","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037531b/a-measure-theoretic-generalization-of-logical-induction",0,"","theory"],["Corrigibility thoughts II: the robot operator","Stuart_Armstrong","2017","blog","LessWrong","www.lesswrong.com/posts/C4Hz3ZPcD4Pef9nfu/corrigibility-thoughts-ii-the-robot-operator",0,"",""],["Corrigibility thoughts III: manipulating versus deceiving","Stuart_Armstrong","2017","blog","LessWrong","www.lesswrong.com/posts/xwT99Ygcnz2hFiqjg/corrigibility-thoughts-iii-manipulating-versus-deceiving",0,"","deception"],["Is it a bias or just a preference? An interesting issue in preference idealization","Caspar Oesterheld","2017","report","casparoesterheld.com","casparoesterheld.com/2017/01/18/is-it-a-bias-or-just-a-preference-an-interesting-issue-in-preference-idealization/",0,"",""],["Decision Theory and the Irrelevance of Impossible Outcomes","Caspar Oesterheld","2017","report","casparoesterheld.com","casparoesterheld.com/2017/01/17/decision-theory-and-the-irrelevance-of-impossible-outcomes/",0,"","theory"],["Agent-Agnostic Human-in-the-Loop Reinforcement Learning","David Abel and 3 others","2017","paper","arXiv preprint","arxiv.org/abs/1701.04079",0,"","deception agents"],["Response to Cegłowski on superintelligence","Matthew Graves","2017","blog","intelligence.org","intelligence.org/2017/01/13/response-to-ceglowski-on-superintelligence/",0,"",""],["Latent Variables and Model Mis-specification","jsteinhardt","2017","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2017/01/10/latent-variables-and-model-mis-specification/",0,"",""],["2016-17 New Year review","Victoria Krakovna","2017","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2017/01/09/2016-17-new-year-review/",0,"",""],["Designing a Safe Autonomous Artificial Intelligence Agent based on Human Self-Regulation","Mark Muraven","2017","paper","arXiv preprint","arxiv.org/abs/1701.01487",0,"","agents governance"],["January 2017 Newsletter","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2017/01/04/january-2017-newsletter/",0,"",""],["Toward negotiable reinforcement learning: shifting priorities in Pareto optimal sequential decision-making","Andrew Critch","2017","paper","arXiv preprint","arxiv.org/abs/1701.01302",0,"","evals policy"],["A Psychoanalytic Approach to the Singularity: Why We Cannot Do Without Auxiliary Constructions","Graham Clarke","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_12",0,"",""],["Agent Foundations for Aligning Machine Intelligence with Human Interests: A Technical Research Agenda","Nate Soares and Benya Fallenstein","2017","report","link.springer.com","link.springer.com/10.1007/978-3-662-54033-6_5",0,"","agents theory"],["Artificial General Intelligence: Timeframes & Policy White Paper","Allison Duettmann","2017","report","foresight.org","foresight.org/publications/AGI-Timeframes&PolicyWhitePaper.pdf",0,"","policy"],["Can the Singularity Be Patented? (And Other IP Conundrums for Converging Technologies)","David Koepsell","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_10",0,"",""],["Computer Simulations as a Technological Singularity in the Empirical Sciences","Juan M. Durán","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_9",0,"",""],["Cyber insurance","Pythagoras Petratos and 2 others","2017","report","doi.org","doi.org/10.1007/978-3-319-09069-6_25",0,"",""],["Diminishing Returns and Recursive Self Improving Artificial Intelligence","Andrew Majot and Roman Yampolskiy","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_7",0,"",""],["Energy, Complexity, and the Singularity","Kent A. Peacock","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_8",0,"",""],["How Change Agencies Can Affect Our Path Towards a Singularity","Ping Zheng and Mohammed-Asif Akhmad","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_4",0,"",""],["Implicitly Assisting Humans to Choose Good Grasps in Robot to Human Handovers","Aaron Bestick and 2 others","2017","report","link.springer.com","link.springer.com/10.1007/978-3-319-50115-4_30",0,"","robustness"],["Introduction to the technological singularity","Stuart Armstrong","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_1",0,"",""],["Learning Robot Objectives from Physical Human Interaction","Andrea Bajcsy and 3 others","2017","report","proceedings.mlr.press","proceedings.mlr.press/v78/bajcsy17a/bajcsy17a.pdf",0,"",""],["Liability For Present And Future Robotics Technology","Trevor N. White and Seth D. Baum","2017","report","academic.oup.com","academic.oup.com/book/2320/chapter-abstract/142464710",0,"",""],["Modeling and interpreting expert disagreement about artificial superintelligence","Seth Baum and 2 others","2017","report","papers.ssrn.com","papers.ssrn.com/sol3/papers.cfm?abstract_id=3104645",0,"",""],["Moral Decision Making Frameworks for Artificial Intelligence","Vincent Conitzer and 4 others","2017","report","moralai.cs.duke.edu","moralai.cs.duke.edu/documents/mai_docs/moralAAAI17.pdf",0,"",""],["New paper: “Optimal polynomial-time estimators”","Rob Bensinger","2017","blog","intelligence.org","intelligence.org/2016/12/31/new-paper-optimal-polynomial-time-estimators/",0,"",""],["Pervasive Spurious Normativity","Gillian K Hadﬁeld and Dylan Hadﬁeld-Menell","2017","report","law.berkeley.edu","www.law.berkeley.edu/files/LET_2017_7.pdf",0,"",""],["Pricing Externalities to Balance Public Risks and Benefits of Research","Sebastian Farquhar and 2 others","2017","report","ncbi.nlm.nih.gov","www.ncbi.nlm.nih.gov/pmc/articles/PMC5576218/pdf/hs.2016.0118.pdf",0,"",""],["Responses to the Journey to the Singularity","Kaj Sotala and Roman Yampolskiy","2017","report","link.springer.com","link.springer.com/10.1007/978-3-662-54033-6_3",0,"",""],["Risk analysis and risk management for the artificial superintelligence research and development process","Anthony M. Barrett and Seth D. Baum","2017","report","gcrinstitute.org","gcrinstitute.org/papers/027_asi-risk.pdf",0,"",""],["Risks of the Journey to the Singularity","Kaj Sotala and Roman Yampolskiy","2017","report","doi.org","doi.org/10.1007/978-3-662-54033-6_2",0,"",""],["Security solutions for intelligent and complex systems","Stuart Armstrong and Roman V. Yampolskiy","2017","report","doi.org","doi.org/10.4018/978-1-5225-0741-3",0,"",""],["Strategic implications of openness in AI development","Nick Bostrom","2017","report","nickbostrom.com","nickbostrom.com/papers/openness.pdf",0,"",""],["Strategic Implications of Openness in AI Development","Nick Bostrom","2017","report","nickbostrom.com","www.nickbostrom.com/papers/openness.pdf",0,"",""],["The Off-Switch Game","Dylan Hadfield-Menell and 3 others","2017","report","ijcai.org","www.ijcai.org/proceedings/2017/32",0,"",""],["The underwriter and the models-solo dances or pas-de-deux? What policy data can tell us about how underwriters use models","Stuart Armstrong and 4 others","2017","report","msamlin.com","www.msamlin.com/content/dam/ms-amlin/corporate/our-world/Whitepapers/MS%20Amlin%20White%20Paper%20The%20underwriter%20and%20the%20models-%20solo%20dances%20or%20pas-de-deux.pdf.downloadasset.pdf",0,"","policy"],["The Wisdom of Nature: An Evolutionary Heuristic for Human Enhancement","Nick Bostrom and Anders Sandberg","2017","report","nickbostrom.com","nickbostrom.com/evolution.pdf",0,"",""],["Transhumanist FAQ 3.0","Nick Bostrom","2017","report","humanityplus.org","www.humanityplus.org/transhumanist-faq?rq=Transhumanist%20FAQ",0,"",""],["Individual Project Fund: Further Details","jsteinhardt","2016","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2016/12/31/individual-project-fund-further-details/",0,"",""],["Pursuing convergent instrumental subgoals on the user's behalf doesn't always require good priors","jessicata","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703752da/pursuing-convergent-instrumental-subgoals-on-the-user-s-behalf-doesn-t-always-require-good-priors",0,"","instrumental-convergence robustness"],["AI Alignment: Why It’s Hard, and Where to Start","Eliezer Yudkowsky","2016","blog","intelligence.org","intelligence.org/2016/12/28/ai-alignment-why-its-hard-and-where-to-start/",0,"",""],["AI Safety Highlights from NIPS 2016","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/12/28/ai-safety-highlights-from-nips-2016/",0,"",""],["Donations for 2016","jsteinhardt","2016","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2016/12/28/donations-for-2016/",0,"",""],["Thinking Outside One’s Paradigm","jsteinhardt","2016","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2016/12/26/thinking-outside-ones-paradigm/",0,"",""],["A Base Camp for Scaling AI","C. J. C. Burges and 6 others","2016","paper","arXiv preprint","arxiv.org/abs/1612.07896",0,"","interpretability deception"],["Faulty Reward Functions in the Wild","Jack Clark and Dario Amodei","2016","report","openai.com","openai.com/blog/faulty-reward-functions/",0,"",""],["Neuro-symbolic EDA-based Optimisation using ILP-enhanced DBNs","Sarmimala Saikia and 5 others","2016","paper","arXiv preprint","arxiv.org/abs/1612.06528",0,"","robustness"],["Extortion and trade negotiations","Stuart_Armstrong","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/RjbTi6ETSo66ygfEY/extortion-and-trade-negotiations",0,"",""],["2016 Expert Survey on Progress in AI","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/2016-expert-survey-on-progress-in-ai/",0,"",""],["Concrete AI tasks for forecasting","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/concrete-ai-tasks-for-forecasting/",0,"","forecasting"],["2016 AI Risk Literature Review and Charity Comparison","Larks","2016","blog","EA Forum","forum.effectivealtruism.org/posts/nSot23sAjoZRgaEwa/2016-ai-risk-literature-review-and-charity-comparison",0,"",""],["December 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/12/13/december-2016-newsletter/",0,"",""],["Experiments in Handwriting with a Neural Network","Distill","2016","report","Distill","distill.pub/2016/handwriting",0,"",""],["Simple and Scalable Predictive Uncertainty Estimation using Deep Ensembles","Balaji Lakshminarayanan and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1612.01474",0,"","evals benchmarks robustness"],["The universal prior is malign","paulfchristiano","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703752a3/the-universal-prior-is-malign",0,"",""],["Joscha Bach on remaining steps to human-level AI","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/joscha-bach-on-the-unfinished-steps-to-human-level-ai/",0,"",""],["Improving Policy Gradient by Exploring Under-appreciated Rewards","Ofir Nachum and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1611.09321",0,"","evals benchmarks policy"],["Predicting HCH using expert advice","jessicata","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037529f/predicting-hch-using-expert-advice",0,"",""],["Thoughts on Updatelessness","Caspar Oesterheld","2016","report","casparoesterheld.com","casparoesterheld.com/2016/11/21/thoughts-on-updatelessness/",0,"",""],["November 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/11/20/november-2016-newsletter/",0,"",""],["Post-fundraiser update","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/11/11/post-fundraiser-update/",0,"",""],["A stochastically verifiable autonomous control architecture with reasoning","Paolo Izzo and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1611.03372",0,"","agents"],["Learning from Untrusted Data","Moses Charikar and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1611.02315",0,"",""],["Neural Architecture Search with Reinforcement Learning","Barret Zoph and Quoc V. Le","2016","paper","arXiv preprint","arxiv.org/abs/1611.01578",0,"",""],["Updatelessness and Son of X","Scott Garrabrant","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037528e/updatelessness-and-son-of-x",0,"","theory"],["Nonparametric General Reinforcement Learning","Jan Leike","2016","report","jan.leike.name","jan.leike.name/publications/Nonparametric%20General%20Reinforcement%20Learning%20-%20Leike%202016.pdf",0,"",""],["Vector-Valued Reinforcement Learning","orthonormal","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375286/vector-valued-reinforcement-learning",0,"",""],["Universal adversarial perturbations","Seyed-Mohsen Moosavi-Dezfooli and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1610.08401",0,"",""],["Artificial Intelligence Safety and Cybersecurity: a Timeline of AI Failures","Roman V. Yampolskiy and M. S. Spellchecker","2016","paper","arXiv preprint","arxiv.org/abs/1610.07997",0,"","forecasting"],["Learning to Protect Communications with Adversarial Neural Cryptography","Martín Abadi and David G. Andersen","2016","paper","arXiv preprint","arxiv.org/abs/1610.06918",0,"","agents"],["White House submissions and report on AI safety","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/10/20/white-house-submissions-and-report-on-ai-safety/",0,"",""],["Transitive negotiations with counterfactual agents","Scott Garrabrant","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375274/transitive-negotiations-with-counterfactual-agents",0,"","agents"],["ALBA on GitHub","Paul Christiano","2016","report","medium.com","medium.com/ai-control/alba-on-github-5636ef510907#.ovfrkun0r",0,"",""],["Deconvolution and Checkerboard Artifacts","Distill","2016","report","Distill","distill.pub/2016/deconv-checkerboard",0,"",""],["OpenAI unconference on machine learning","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/10/15/openai-unconference-on-machine-learning/",0,"",""],["How to Use t-SNE Effectively","Distill","2016","report","Distill","distill.pub/2016/misread-tsne",0,"",""],["MIRI AMA, and a talk on logical induction","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/10/11/miri-ama-and-a-talk-on-logical-induction/",0,"","theory"],["October 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/10/09/october-2016-newsletter/",0,"",""],["Situational Awareness by Risk-Conscious Skills","Daniel J. Mankowitz","2016","paper","arXiv preprint","arxiv.org/abs/1610.02847",0,"","situational-awareness"],["A Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks","Dan Hendrycks","2016","paper","International Conference on Learning Representations 2017","arxiv.org/abs/1610.02136",0,"",""],["CSRBAI talks on agent models and multi-agent dilemmas","Alex Vermeer","2016","blog","intelligence.org","intelligence.org/2016/10/06/csrbai-talks-agent-models/",0,"","agents"],["Xception: Deep Learning with Depthwise Separable Convolutions","François Chollet","2016","paper","arXiv preprint","arxiv.org/abs/1610.02357",0,"","robustness"],["Logical inductor limits are dense under pointwise convergence","SamEisenstat","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037525d/logical-inductor-limits-are-dense-under-pointwise-convergence",0,"",""],["Capability amplification","Paul Christiano","2016","report","medium.com","medium.com/ai-control/policy-amplification-6a70cbee4f34#.31incu10a",0,"","scalable-oversight"],["Backup utility functions as a fail-safe AI technique","Caspar Oesterheld","2016","report","longtermrisk.org","longtermrisk.org/files/backup-utility-functions.pdf",0,"",""],["Information gathering actions over human internal state","Dorsa Sadigh and 3 others","2016","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/7759036/",0,"",""],["Looking back at my grad school journey","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/09/30/looking-back-at-my-grad-school-journey/",0,"",""],["The set of Logical Inductors is not Convex","Scott Garrabrant","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375250/the-set-of-logical-inductors-is-not-convex",0,"","theory"],["UbuntuWorld 1.0 LTS - A Platform for Automated Problem Solving & Troubleshooting in the Ubuntu OS","Tathagata Chakraborti and 4 others","2016","paper","arXiv preprint","arxiv.org/abs/1609.08524",0,"","evals agents"],["Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","Yonghui Wu and 30 others","2016","paper","arXiv preprint","arxiv.org/abs/1609.08144",0,"","evals benchmarks robustness"],["Would You Hand Over a Decision to a Machine?","Seán Ó hÉigeartaigh","2016","report","papers.ssrn.com","papers.ssrn.com/abstract=3446679",0,"",""],["Logical Inductors that trust their limits","Scott Garrabrant","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037523a/logical-inductors-that-trust-their-limits",0,"","theory"],["Superintelligence FAQ","Scott Alexander","2016","blog","LessWrong","www.lesswrong.com/posts/LTtNXM9shNM9AC2mp/superintelligence-faq",0,"",""],["A Formal Solution to the Grain of Truth Problem","Jan Leike and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1609.05058",0,"","agents"],["Exploration Potential","Jan Leike","2016","paper","arXiv preprint","arxiv.org/abs/1609.04994",0,"","agents"],["Long-Term Trends in the Public Perception of Artificial Intelligence","Ethan Fast and Eric Horvitz","2016","paper","arXiv preprint","arxiv.org/abs/1609.04904",0,"","evals"],["MIRI’s 2016 Fundraiser","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/09/16/miris-2016-fundraiser/",0,"",""],["(C)IRL is not solely a learning process","Stuart_Armstrong","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037520e/c-irl-is-not-solely-a-learning-process",0,"",""],["Graph Aggregation","Ulle Endriss and Umberto Grandi","2016","paper","Artificial Intelligence, Volume 245, pages 86-114, 2017","arxiv.org/abs/1609.03765",0,"","deception robustness"],["New paper: “Logical induction”","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/09/12/new-paper-logical-induction/",0,"","theory"],["Logical Induction","Scott Garrabrant and 4 others","2016","paper","arXiv preprint","arxiv.org/abs/1609.03543",0,"","evals robustness theory"],["Attention and Augmented Recurrent Neural Networks","Distill","2016","report","Distill","distill.pub/2016/augmented-rnns",0,"",""],["Conversation with Tom Griffiths","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/conversation-with-tom-griffiths/",0,"",""],["Tom Griffiths on Cognitive Science and AI","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/tom-griffiths-on-cognitive-science-and-ai/",0,"",""],["Grant announcement from the Open Philanthropy Project","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/09/06/grant-open-philanthropy/",0,"",""],["September 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/09/03/september-2016-newsletter/",0,"",""],["Sources of advantage for digital agents over biological agents","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/sources-of-advantage-for-artificial-intelligence/",0,"","agents"],["What if you turned the world’s hardware into AI minds?","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/what-if-you-turned-the-worlds-hardware-into-ai-minds/",0,"",""],["Formalizing preference utilitarianism in physical world models","Caspar Oesterheld","2016","report","doi.org","doi.org/10.1007/s11229-015-0883-1",0,"",""],["CSRBAI talks on preference specification","Alex Vermeer","2016","blog","intelligence.org","intelligence.org/2016/08/30/csrbai-talks-preference-specification/",0,"",""],["Why does deep and cheap learning work so well?","Henry W. Lin and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1608.08225",0,"",""],["Highlights from the Deep Learning Summer School","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/08/25/highlights-from-the-deep-learning-summer-school/",0,"",""],["Two Strange Facts","jsteinhardt","2016","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2016/08/25/two-strange-facts/",0,"",""],["Wireheading Done Right: Stay Positive Without Going Insane","Andres Gomez Emilsson","2016","report","qri.org","qri.org/pdf/wireheading-done-right.pdf",0,"","reward-hacking"],["Modeling the capabilities of advanced AI systems as episodic reinforcement learning","jessicata","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703751eb/modeling-the-capabilities-of-advanced-ai-systems-as-episodic-reinforcement-learning",0,"",""],["Deepmind Plans for Rat-Level AI","moridinamael","2016","blog","LessWrong","www.lesswrong.com/posts/SSj6Rrx9ZN9WF6eaB/deepmind-plans-for-rat-level-ai",0,"","forecasting"],["Superintelligence via whole brain emulation","AlexMennen","2016","blog","LessWrong","www.lesswrong.com/posts/jMKZKc2GiFGegRXvN/superintelligence-via-whole-brain-emulation",0,"",""],["Examples of early action on risks","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/examples-of-early-action-on-a-risk/",0,"",""],["Towards Evaluating the Robustness of Neural Networks","Nicholas Carlini   David Wagner","2016","paper","arXiv preprint","arxiv.org/abs/1608.04644",0,"","evals benchmarks robustness"],["CSRBAI talks on robustness and error-tolerance","Alex Vermeer","2016","blog","intelligence.org","intelligence.org/2016/08/15/csrbai-talks-on-robustness-and-error-tolerance/",0,"","robustness"],["A General Safety Framework for Learning-Based Control in Uncertain Robotic Systems.","Jaime F and 9 others","2016","paper","arXiv preprint","arxiv.org/abs/1705.01292",0,"","policy"],["Generating Plans that Predict Themselves.","Jaime F and 12 others","2016","paper","Jaime F. Fisac, Chang Liu, Jessica B. Hamrick, S. Shankar Sastry,\n  J. Karl Hedrick, Thomas L. Griffiths, and Anca D. Dragan. \"Generating Plans\n  that Predict Themselves\". Workshop on Algorithmic Foundations of Robotics\n  (WAFR), 2016","arxiv.org/abs/1802.05250",0,"",""],["Inferring and Assisting with Constraints in Shared Autonomy.","Negar Mehr and 2 others","2016","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/7799299",0,"",""],["Optimal Polynomial-Time Estimators: A Bayesian Notion of Approximation Algorithm","Vanessa Kosoy and Alexander Appel","2016","paper","arXiv preprint","arxiv.org/abs/1608.04112",0,"",""],["Implicitly Assisting Humans to Choose Good Grasps in Robot to Human Handovers.","Aaron Bestick and 2 others","2016","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~anca/papers/ISER16_influence.pdf",0,"","robustness"],["Information Gathering Actions Over Human Internal State.","Dorsa Sadigh and 5 others","2016","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~anca/papers/IROS16_active.pdf",0,"",""],["MDPs with Unawareness in Robotics.","Nan Rong and 3 others","2016","report","auai.org","auai.org/uai2016/proceedings/papers/294.pdf",0,"",""],["Planning for Autonomous Cars that Leverage Effects on Human Actions.","Dorsa Sadigh and 3 others","2016","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~sastry/pubs/Pdfs%20of%202016/SadighPlanning2016.pdf",0,"",""],["Sufficient Conditions for Causality to be Transitive.","Joseph Y and Halpern","2016","report","cs.cornell.edu","www.cs.cornell.edu/home/halpern/papers/transitivity.pdf",0,"",""],["Friendly AI as a global public good","Michael Wulfsohn","2016","blog","aiimpacts.org","aiimpacts.org/friendly-ai-as-a-global-public-good/",0,"","robustness"],["Andrew Critch: Logical induction — progress in AI alignment","EA Global","2016","blog","EA Forum","forum.effectivealtruism.org/posts/HYHaBsukLkoE72zTd/andrew-critch-logical-induction-progress-in-ai-alignment",0,"","theory"],["Max Tegmark: Risks and benefits of advanced artificial intelligence","EA Global","2016","blog","EA Forum","forum.effectivealtruism.org/posts/7Y5BffB9scQdord5N/max-tegmark-risks-and-benefits-of-advanced-artificial",0,"",""],["MIRI strategy update: 2016","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/08/05/miri-strategy-update-2016/",0,"",""],["August 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/08/03/august-2016-newsletter/",0,"",""],["Costs of extinction risk mitigation","Michael Wulfsohn","2016","blog","aiimpacts.org","aiimpacts.org/costs-of-extinction-risk-mitigation/",0,"",""],["2016 summer program recap","Alex Vermeer","2016","blog","intelligence.org","intelligence.org/2016/08/02/2016-summer-program-recap/",0,"",""],["Clopen AI: Openness in different aspects of AI development","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/08/01/clopen-ai-openness-in-different-aspects-of-ai-development/",0,"",""],["2015 in review","Malo Bourgon","2016","blog","intelligence.org","intelligence.org/2016/07/29/2015-in-review/",0,"",""],["Do Artificial Reinforcement-Learning Agents Matter Morally?","Brian Tomasik","2016","report","longtermrisk.org","longtermrisk.org/do-artificial-reinforcement-learning-agents-matter-morally/",0,"","agents"],["Mammalian Value Systems","Gopal P. Sarma and Nick J. Hay","2016","paper","Informatica Vol. 41 No. 3 (2017)","arxiv.org/abs/1607.08289",0,"","agents"],["New paper: “Alignment for advanced machine learning systems”","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/07/27/alignment-machine-learning/",0,"",""],["A Model of Pathways to Artificial Superintelligence Catastrophe for Risk and Decision Analysis","Anthony M. Barrett and Seth D. Baum","2016","paper","arXiv preprint","arxiv.org/abs/1607.07730",0,"","evals"],["Submission to the OSTP on AI outcomes","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/07/23/ostp/",0,"",""],["Predicting Enemy's Actions Improves Commander Decision-Making","Michael Ownby and Alexander Kott","2016","paper","arXiv preprint","arxiv.org/abs/1607.06759",0,"","evals deception"],["Layer Normalization","Jimmy Lei Ba","2016","paper","arXiv preprint","arxiv.org/abs/1607.06450",0,"",""],["Three Oracle designs","Stuart_Armstrong","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703751d1/three-oracle-designs",0,"",""],["Exploiting Vagueness for Multi-Agent Consensus","Michael Crosscombe and Jonathan Lawry","2016","paper","arXiv preprint","arxiv.org/abs/1607.05540",0,"","deception agents"],["The Naive Utility Calculus: Computational Principles Underlying Commonsense Psychology","Julian Jara-Ettinger and 3 others","2016","report","cell.com","www.cell.com/trends/cognitive-sciences/fulltext/S1364-6613(16)30053-5?_returnURL=https%3A%2F%2Flinkinghub.elsevier.com%2Fretrieve%2Fpii%2FS1364661316300535%3Fshowall%3Dtrue",0,"","deception"],["Adversarial examples in the physical world","Alexey Kurakin","2016","paper","arXiv preprint","arxiv.org/abs/1607.02533",0,"","deception robustness"],["July 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/07/05/july-2016-newsletter/",0,"",""],["Returns to scale in research","Michael Wulfsohn","2016","blog","aiimpacts.org","aiimpacts.org/returns-to-scale-in-research/",0,"",""],["A Hybrid POMDP-BDI Agent Architecture with Online Stochastic Planning and Plan Caching","Gavin Rens and Deshendran Moodley","2016","paper","arXiv preprint","arxiv.org/abs/1607.00656",0,"","evals agents"],["The Unilateralist’s Curse and the Case for a Principle of Conformity","Nick Bostrom and 2 others","2016","report","ncbi.nlm.nih.gov","www.ncbi.nlm.nih.gov/pmc/articles/PMC4959137/",0,"",""],["New paper: “A formal solution to the grain of truth problem”","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/06/30/grain-of-truth/",0,"",""],["Towards A Virtual Assistant That Can Be Taught New Tasks In Any Domain By Its End-Users","I. Dan Melamed and Nobal B. Niraula","2016","paper","arXiv preprint","arxiv.org/abs/1607.00061",0,"","agents"],["Selected Citations","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/selected-citations/",0,"",""],["Bridging Nonlinearities and Stochastic Regularizers with Gaussian Error Linear Units","Dan Hendrycks and Kevin Gimpel","2016","paper","arXiv preprint","arxiv.org/abs/1606.08415",0,"","evals"],["Towards Verified Artificial Intelligence","Sanjit A. Seshia and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1606.08514",0,"","assurance"],["Artificial Fun: Mapping Minds to the Space of Fun","Soenke Ziesche and Roman V. Yampolskiy","2016","paper","arXiv preprint","arxiv.org/abs/1606.07092",0,"",""],["New AI safety research agenda from Google Brain","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/06/22/new-ai-safety-research-agenda-from-google-brain/",0,"",""],["Visualizing Dynamics: from t-SNE to SEMI-MDPs","","2016","paper","arXiv preprint","arxiv.org/abs/1606.07112",0,"","evals agents policy robustness"],["Clustering with a Reject Option: Interactive Clustering as Bayesian Prior Elicitation","Akash Srivastava and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1606.05896",0,"","robustness"],["Cooperative Inverse Reinforcement Learning vs. Irrational Human Preferences","orthonormal","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703751bf/cooperative-inverse-reinforcement-learning-vs-irrational-human-preferences",0,"",""],["Avoiding Imposters and Delinquents: Adversarial Crowdsourcing and Peer Prediction","Jacob Steinhardt and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1606.05374",0,"","evals"],["Increasing the Interpretability of Recurrent Neural Networks Using Hidden Markov Models","","2016","paper","arXiv preprint","arxiv.org/abs/1606.05320",0,"","interpretability robustness"],["Unsupervised Risk Estimation Using Only Conditional Independence Structure","Jacob Steinhardt","2016","paper","arXiv preprint","arxiv.org/abs/1606.05313",0,"",""],["June 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/06/12/june-2016-newsletter/",0,"",""],["In memoryless Cartesian environments, every UDT policy is a CDT+SIA policy","jessicata","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703751b2/in-memoryless-cartesian-environments-every-udt-policy-is-a",0,"","policy theory"],["Generative Adversarial Imitation Learning","Jonathan Ho","2016","paper","arXiv preprint","arxiv.org/abs/1606.03476",0,"",""],["The Mythos of Model Interpretability","Zachary C. Lipton","2016","paper","arXiv preprint","arxiv.org/abs/1606.03490",0,"","interpretability deception"],["Learning Language Games through Interaction","Sida I. Wang   Percy Liang   Christopher D. Manning","2016","paper","arXiv preprint","arxiv.org/abs/1606.02447",0,"",""],["Fundamental lssues Of Artificial Intelligence","Vincent C. Müller","2016","report","goodreads.com","www.goodreads.com/book/show/30528889-fundamental-issues-of-artificial-intelligence",0,"",""],["OpenAI Gym","Greg Brockman and 6 others","2016","paper","arXiv preprint","arxiv.org/abs/1606.01540",0,"","benchmarks"],["New paper: “Safely interruptible agents”","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/06/01/new-paper-safely-interruptible-agents/",0,"","agents"],["Suffering-focused AI safety: Why “fail-safe” measures might be our top intervention","Lukas Gloor","2016","report","foundational-research.org","foundational-research.org/files/suffering-focused-ai-safety.pdf",0,"",""],["Synthesizing the preferred inputs for neurons in neural networks via deep generator networks","Anh Nguyen","2016","paper","arXiv preprint","arxiv.org/abs/1605.09304",0,"","interpretability"],["Learning with catastrophes","Paul Christiano","2016","report","medium.com","medium.com/ai-control/learning-with-catastrophes-59387b55cc30#.ek0ew1n77",0,"",""],["Transparency reports make AI decision-making accountable","CMU","2016","report","phys.org","phys.org/news/2016-05-transparency-ai-decision-making-accountable.html",0,"","interpretability"],["Stabilizing logical counterfactuals by pseudorandomization","Vanessa Kosoy","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703750cf/stabilizing-logical-counterfactuals-by-pseudorandomization",0,"",""],["Quantifying the Far Future Effects of Interventions","MichaelDickens","2016","blog","EA Forum","forum.effectivealtruism.org/posts/Gg6SvNy8ZRAjYRbCZ/quantifying-the-far-future-effects-of-interventions",0,"",""],["Error in Armstrong and Sotala 2012","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/error-in-armstrong-and-sotala-2012/",0,"",""],["May 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/05/13/may-2016-newsletter/",0,"",""],["Metasurvey: predict the predictors","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/metasurvey-predict-the-predictors/",0,"",""],["Avoiding Wireheading with Value Reinforcement Learning","Tom Everitt and Marcus Hutter","2016","paper","arXiv preprint","arxiv.org/abs/1605.03143",0,"","reward-hacking agents robustness"],["Self-Modification of Policy and Utility Function in Rational Agents","Tom Everitt and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1605.03142",0,"","evals agents policy"],["Potential Risks from Advanced Artificial Intelligence: The Philanthropic Opportunity","Holden Karnofsky","2016","blog","EA Forum","forum.effectivealtruism.org/posts/T7e42LXSDCkoF33Mt/potential-risks-from-advanced-artificial-intelligence-the",0,"",""],["A new MIRI research program with a machine learning focus","admin","2016","blog","intelligence.org","intelligence.org/2016/05/04/announcing-a-new-research-program/",0,"",""],["You Say You Want Transparency and Interpretability?","Rayid Ghani","2016","report","rayidghani.com","www.rayidghani.com/you-say-you-want-transparency-and-interpretability",0,"","interpretability"],["Global Catastrophic Risks 2016","Owen Cotton-Barratt and 4 others","2016","report","globalprioritiesproject.org","globalprioritiesproject.org/2016/04/global-catastrophic-risks-2016/",0,"",""],["Classifying Options for Deep Reinforcement Learning","Kai Arulkumaran and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1604.08153",0,"","policy"],["Limits to Verification and Validation of Agentic Behavior","David J. Jilk","2016","paper","arXiv preprint","arxiv.org/abs/1604.06963",0,"","agents governance"],["New papers dividing logical uncertainty into two subproblems","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/04/21/two-new-papers-uniform/",0,"",""],["Asymptotic Convergence in Online Learning with Unbounded Delays","Scott Garrabrant and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1604.05280",0,"","evals forecasting robustness"],["Inductive Coherence","Scott Garrabrant and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1604.05288",0,"",""],["An artificial intelligence tool for heterogeneous team formation in the classroom","Juan M. Alberola and 4 others","2016","paper","Knowledge-Based Systems, 2016","arxiv.org/abs/1604.04721",0,"",""],["Using humility to counteract shame","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/04/15/using-humility-to-counteract-shame/",0,"",""],["Moving Beyond the Turing Test with the Allen AI Science Challenge","Carissa Schoenick and 4 others","2016","paper","arXiv preprint","arxiv.org/abs/1604.04315",0,"",""],["The many counterfactuals of counterfactual mugging","Scott Garrabrant","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375158/the-many-counterfactuals-of-counterfactual-mugging",0,"",""],["April 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/04/11/april-2016-newsletter/",0,"",""],["The AGI Containment Problem","James Babcock and 2 others","2016","paper","Lecture Notes in Artificial Intelligence 9782 (AGI 2016,\n  Proceedings) 53-63","arxiv.org/abs/1604.00545",0,"",""],["New paper on bounded Löb and robust cooperation of bounded agents","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/03/31/new-paper-on-bounded-lob/",0,"","agents"],["The Age of Em: Work, Love and Life when Robots Rule the Earth","Robin Hanson","2016","report","goodreads.com","www.goodreads.com/book/show/26831944-the-age-of-em",0,"",""],["MIRI has a new COO: Malo Bourgon","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/03/30/miri-has-a-new-coo-malo-bourgon/",0,"",""],["Concrete AI tasks bleg","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/concrete-ai-tasks-bleg/",0,"",""],["Announcing a new colloquium series and fellows program","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/03/28/announcing-a-new-colloquium-series-and-fellows-program/",0,"",""],["So Far: Unfriendly AI Edition","Eliezer Yudkowsky","2016","report","econlib.org","www.econlib.org/archives/2016/03/so_far_unfriend.html",0,"",""],["Crystal Society trilogy: Inside the mind of an AI","Max Harms","2016","report","crystal.raelifin.com","crystal.raelifin.com/",0,"",""],["Seeking Research Fellows in Type Theory and Machine Self-Reference","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/03/18/seeking-research-fellows-in-type-theory-and-machine-self-reference/",0,"",""],["A Signaling Game Approach to Databases Querying and Interaction","Ben McCamish and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1603.04068",0,"","agents"],["Mysteries of global hardware","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/mysteries-of-global-hardware/",0,"",""],["March 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/03/05/march-2016-newsletter/",0,"",""],["John Horgan interviews Eliezer Yudkowsky","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/03/02/john-horgan-interviews-eliezer-yudkowsky/",0,"",""],["Guided Cost Learning: Deep Inverse Optimal Control via Policy Optimization","Chelsea Finn and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1603.00448",0,"","evals policy"],["New paper: “Defining human values for value learners”","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/02/29/new-paper-defining-human-values-for-value-learners/",0,"",""],["The informed oversight problem","Paul Christiano","2016","report","medium.com","medium.com/ai-control/the-informed-oversight-problem-1b51b4f66b35#.ddvq5rheo",0,"",""],["Introductory resources on AI safety research","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/02/28/introductory-resources-on-ai-safety-research/",0,"",""],["Toy model: convergent instrumental goals","Stuart_Armstrong","2016","blog","LessWrong","www.lesswrong.com/posts/htCBHjWNvzScEhFoA/toy-model-convergent-instrumental-goals",0,"","instrumental-convergence"],["ALBA: An explicit proposal for aligned AI","Paul Christiano","2016","report","medium.com","medium.com/ai-control/alba-an-explicit-proposal-for-aligned-ai-17a55f60bbcf#.3jwpm81j8",0,"",""],["Speculations on information under logical uncertainty","TsviBT","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703750e8/speculations-on-information-under-logical-uncertainty",0,"",""],["Latent Skill Embedding for Personalized Lesson Sequence Recommendation","Siddharth Reddy and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1602.07029",0,"","evals benchmarks robustness"],["The Singularity May Never Be Near","Toby Walsh","2016","paper","arXiv preprint","arxiv.org/abs/1602.06462",0,"",""],["Global computing capacity","Katja Grace","2016","blog","aiimpacts.org","aiimpacts.org/global-computing-capacity/",0,"",""],["\"Why Should I Trust You?\": Explaining the Predictions of Any Classifier","Marco Tulio Ribeiro and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1602.04938",0,"","interpretability"],["Bayesian Optimization with Safety Constraints: Safe and Automatic Parameter Tuning in Robotics","Felix Berkenkamp1 and 2 others","2016","paper","arXiv preprint","arxiv.org/abs/1602.04450",0,"","evals"],["Designing Intelligent Instruments","Kevin H. Knuth and 2 others","2016","paper","AIP Conference Proceedings 954, American Institute of Physics,\n  Melville NY, 203-211, 2007","arxiv.org/abs/1602.04290",0,"","deception"],["Energetics of the brain and AI","Anders Sandberg","2016","paper","arXiv preprint","arxiv.org/abs/1602.04019",0,"",""],["Parametric Bounded Löb's Theorem and Robust Cooperation of Bounded Agents","Andrew Critch","2016","paper","arXiv preprint","arxiv.org/abs/1602.04184",0,"","evals agents"],["Modeling Human Ad Hoc Coordination","Peter M. Krafft and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1602.03924",0,"","agents"],["Research Priorities for Robust and Beneficial Artificial Intelligence","Stuart Russell and 2 others","2016","paper","AI Magazine 36:4 (2015)","arxiv.org/abs/1602.03506",0,"",""],["Graying the black box: Understanding DQNs","\\name","2016","paper","arXiv preprint","arxiv.org/abs/1602.02658",0,"",""],["Practical Black-Box Attacks against Deep Learning Systems using Adversarial Examples","","2016","paper","arXiv preprint","arxiv.org/abs/1602.02697",0,"","evals robustness training-data"],["February 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/02/06/february-2016-newsletter/",0,"",""],["Mastering the game of Go with deep neural networks and tree search","David Silver and 19 others","2016","report","nature.com","www.nature.com/nature/journal/v529/n7587/full/nature16961.html",0,"",""],["A survey of research questions for robust and beneficial AI","Daniel Dewey and 2 others","2016","report","futureoflife.org","futureoflife.org/data/documents/research_survey.pdf?x96845",0,"",""],["A survey of research questions for robust and beneficial AI","Future of Life Institute","2016","report","futureoflife.org","futureoflife.org/data/documents/research_survey.pdf",0,"",""],["Towards Resolving Unidentifiability in Inverse Reinforcement Learning","Kareem Amin","2016","paper","arXiv preprint","arxiv.org/abs/1601.06569",0,"","agents"],["The Singularity Controversy, Part I: Lessons Learned and Open Questions: Conclusions from the Battle on the Legitimacy of the Debate","Amnon H. Eden","2016","paper","arXiv preprint","arxiv.org/abs/1601.05977",0,"","policy"],["Coordinated human action as example of superhuman intelligence","Ben Hoffman","2016","blog","aiimpacts.org","aiimpacts.org/coordinated-human-action-example-superhuman-intelligence/",0,"",""],["To contribute to AI safety, consider doing AI research","Victoria Krakovna","2016","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2016/01/16/to-contribute-to-ai-safety-consider-doing-ai-research/",0,"",""],["To contribute to AI safety, consider doing AI research","Vika","2016","blog","LessWrong","www.lesswrong.com/posts/pCesigb4NjzvoNKWB/to-contribute-to-ai-safety-consider-doing-ai-research",0,"",""],["The correct response to uncertainty is *not* half-speed","AnnaSalamon","2016","blog","LessWrong","www.lesswrong.com/posts/FMkQtPvzsriQAow5q/the-correct-response-to-uncertainty-is-not-half-speed",0,"","theory"],["Analysis of Algorithms and Partial Algorithms","Andrew MacFie","2016","paper","Artificial General Intelligence 2016, New York, USA, July 16-19,\n  2016, Proceedings, 284-293","arxiv.org/abs/1601.03411",0,"","deception agents"],["Difficulty of Predicting the Maximum of Gaussians","jsteinhardt","2016","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2016/01/13/difficulty-of-predicting-the-maximum-of-gaussians/",0,"",""],["End-of-the-year fundraiser and grant successes","Nate Soares","2016","blog","intelligence.org","intelligence.org/2016/01/12/end-of-the-year-fundraiser-and-grant-successes/",0,"",""],["Another view of quantilizers: avoiding Goodhart's Law","jessicata","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703750b1/another-view-of-quantilizers-avoiding-goodhart-s-law",0,"","goodharts-law"],["Logical counterfactuals for random algorithms","Vanessa Kosoy","2016","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf06703750a5/logical-counterfactuals-for-random-algorithms",0,"",""],["January 2016 Newsletter","Rob Bensinger","2016","blog","intelligence.org","intelligence.org/2016/01/03/january-2016-newsletter/",0,"",""],["Agential Risks: A Comprehensive Introduction","Phil Torres","2016","report","jetpress.org","jetpress.org/v26.2/torres.pdf",0,"","agents"],["Building Machines That Learn and Think Like People","Brenden M. Lake and 3 others","2016","paper","arXiv preprint","arxiv.org/abs/1604.00289",0,"","agents"],["Defining human values for value learners","Kaj Sotala","2016","report","intelligence.org","intelligence.org/files/DefiningValuesForValueLearners.pdf",0,"",""],["Embedding Ethical Principles in Collective Decision Support Systems","Joshua Greene and 4 others","2016","report","projects.iq.harvard.edu","projects.iq.harvard.edu/files/mcl/files/greene-et-al-ethical-principles-machines-aaai16.pdf",0,"",""],["Formalizing convergent instrumental goals","Tsvi Benson-Tilsen and Nate Soares","2016","report","intelligence.org","intelligence.org/files/FormalizingConvergentGoals.pdf",0,"","instrumental-convergence"],["Future progress in artificial intelligence: A survey of expert opinion","Vincent C. Müller and Nick Bostrom","2016","report","nickbostrom.com","nickbostrom.com/papers/survey.pdf",0,"",""],["Growing Recursive Self-Improvers","Bas R. Steunebrink and 2 others","2016","report","link.springer.com","link.springer.com/10.1007/978-3-319-41649-6_13",0,"",""],["How the Simulation Argument Dampens Future Fanaticism","Brian Tomasik","2016","report","longtermrisk.org","longtermrisk.org/how-the-simulation-argument-dampens-future-fanaticism",0,"",""],["Learning the Preferences of Ignorant, Inconsistent Agents","Owain Evans and 2 others","2016","report","stuhlmueller.org","stuhlmueller.org/papers/preferences-aaai2016.pdf",0,"","agents"],["Planning for Autonomous Cars that Leverage Effects on Human Actions","Dorsa Sadigh and 3 others","2016","report","roboticsproceedings.org","www.roboticsproceedings.org/rss12/p29.pdf",0,"",""],["Policy desiderata in the development of machine superintelligence","Nick Bostrom and 2 others","2016","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Policy-Desiderata-in-the-Development-of-Machine-Superintelligence.pdf",0,"","policy"],["Probabilistic Models of Cognition","Noah D. Goodman and Joshua B. Tenenbaum","2016","report","probmods.org","probmods.org/",0,"",""],["Quantilizers: A safer alternative to maximizers for limited optimization","Jessica Taylor","2016","report","intelligence.org","intelligence.org/files/QuantilizersSaferAlternative.pdf",0,"",""],["Racing to the precipice: a model of artificial intelligence development","Stuart Armstrong and 2 others","2016","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Racing-to-the-precipice-a-model-of-artificial-intelligence-development.pdf",0,"",""],["Rationality and Intelligence: A Brief Update","Stuart Russell","2016","report","link.springer.com","link.springer.com/10.1007/978-3-319-26485-1_2",0,"",""],["Robots in war: the next weapons of mass destruction?","Stuart Russell","2016","report","weforum.org","www.weforum.org/agenda/2016/01/robots-in-war-the-next-weapons-of-mass-destruction/",0,"",""],["Safely Interruptible Agents","Laurent Orseau and Stuart Armstrong","2016","report","auai.org","auai.org/uai2016/proceedings/papers/68.pdf",0,"","agents"],["Suffering-focused AI safety: Why \"fail-safe'\" measures might be our top intervention","Lukas Gloor","2016","report","longtermrisk.org","longtermrisk.org/files/suffering-focused-ai-safety.pdf",0,"",""],["The Control Problem. Excerpts from Superintelligence: Paths, Dangers, Strategies","Nick Bostrom","2016","report","doi.org","doi.org/10.1002/9781118922590.ch23",0,"",""],["The Liability Problem for Autonomous Artificial Agents","Peter M Asaro","2016","report","peterasaro.org","peterasaro.org/writing/Asaro,%20Ethics%20Auto%20Agents,%20AAAI.pdf",0,"","agents"],["The Technological Singularity: Managing the Journey","Roman Yampolskiy and Stuart Armstrong","2016","report","link.springer.com","link.springer.com/book/10.1007/978-3-662-54033-6",0,"",""],["Towards interactive inverse reinforcement learning","Stuart Armstrong and Jan Leike","2016","report","jan.leike.name","jan.leike.name/publications/Towards%20Interactive%20Inverse%20Reinforcement%20Learning%20-%20Armstrong,%20Leike%202016.pdf",0,"",""],["2015-16 New Year review","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/12/31/2015-16-new-year-review/",0,"",""],["Safety engineering, target selection, and alignment theory","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/12/31/safety-engineering-target-selection-and-alignment-theory/",0,"",""],["Highlights and impressions from NIPS conference on machine learning","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/12/24/highlights-and-impressions-from-nips-conference-on-machine-learning/",0,"",""],["Multi-Level Cause-Effect Systems","Krzysztof Chalupka and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1512.07942",0,"","robustness"],["Toward a Research Agenda in Adversarial Reasoning: Computational Approaches to Anticipating the Opponent's Intent and Actions","Alexander Kott and Michael Ownby","2015","paper","arXiv preprint","arxiv.org/abs/1512.07943",0,"","deception"],["The need to scale MIRI’s methods","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/12/23/need-scale-miris-methods/",0,"",""],["A sketch of a value-learning sovereign","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375072/a-sketch-of-a-value-learning-sovereign",0,"",""],["Learning the Preferences of Ignorant, Inconsistent Agents","Owain Evans and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1512.05832",0,"","agents"],["Modeling Progress in AI","Miles Brundage","2015","paper","arXiv preprint","arxiv.org/abs/1512.05849",0,"","evals"],["Some work on connecting UDT and Reinforcement Learning","IAFF-User-111","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037508c/some-work-on-connecting-udt-and-reinforcement-learning",0,"",""],["Jed McCaleb on Why MIRI Matters","Guest","2015","blog","intelligence.org","intelligence.org/2015/12/15/jed-mccaleb-on-why-miri-matters/",0,"",""],["Logical Counterfactuals Consistent Under Self-Modification","abramdemski","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375085/logical-counterfactuals-consistent-under-self-modification",0,"",""],["Saying 'AI safety research is a Pascal's Mugging' isn't a strong response","Robert_Wiblin","2015","blog","EA Forum","forum.effectivealtruism.org/posts/vYb2qEyqv76L62izD/saying-ai-safety-research-is-a-pascal-s-mugging-isn-t-a",0,"",""],["The Rationale behind the Concept of Goal","Guido Governatori and 4 others","2015","paper","Theory and Practice of Logic Programming 16 (2016) 296-324","arxiv.org/abs/1512.04021",0,"","agents"],["OpenAI and other news","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/12/11/openai-and-other-news/",0,"",""],["Human-level concept learning through probabilistic program induction","Brenden M. Lake and 2 others","2015","report","cs.cmu.edu","www.cs.cmu.edu/~rsalakhu/papers/LakeEtAl2015Science.pdf",0,"",""],["Deep Residual Learning for Image Recognition","Kaiming He   Xiangyu Zhang   Shaoqing Ren   Jian Sun","2015","paper","arXiv preprint","arxiv.org/abs/1512.03385",0,"","evals"],["Deep Speech 2: End-to-End Speech Recognition in English and Mandarin","Dario Amodei and 33 others","2015","paper","arXiv preprint","arxiv.org/abs/1512.02595",0,"","benchmarks"],["New paper: “Proof-producing reflection for HOL”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/12/04/new-paper-proof-producing-reflection-for-hol/",0,"",""],["Reward engineering","Paul Christiano","2015","report","medium.com","medium.com/ai-control/reward-engineering-f8b5de40d075",0,"",""],["December 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/12/03/december-2015-newsletter/",0,"",""],["MIRI’s 2015 Winter Fundraiser!","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/12/01/miri-2015-winter-fundraiser/",0,"",""],["New paper: “Quantilizers”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/11/29/new-paper-quantilizers/",0,"",""],["Risks from general artificial intelligence without an intelligence explosion","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/11/29/ai-risk-without-an-intelligence-explosion/",0,"",""],["New paper: “Formalizing convergent instrumental goals”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/11/26/new-paper-formalizing-convergent-instrumental-goals/",0,"","instrumental-convergence"],["A Roadmap towards Machine Intelligence","Tomas Mikolov","2015","paper","arXiv preprint","arxiv.org/abs/1511.08130",0,"",""],["Convergent Learning: Do different neural networks learn the same representations?","Yixuan Li","2015","paper","arXiv preprint","arxiv.org/abs/1511.07543",0,"","interpretability"],["Recently at AI Impacts","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/recently-at-ai-impacts/",0,"",""],["November 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/11/03/november-2015-newsletter/",0,"",""],["Superrationality in arbitrary games","Vanessa Kosoy","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375058/superrationality-in-arbitrary-games",0,"",""],["Edge.org contributors discuss the future of AI","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/11/01/edge-org-contributors-discuss-the-future-of-ai/",0,"",""],["Working at EA organizations series: Machine Intelligence Research Institute","SoerenMind","2015","blog","EA Forum","forum.effectivealtruism.org/posts/WfNZoquLhRnT3nC4e/working-at-ea-organizations-series-machine-intelligence",0,"",""],["Glossary of AI Risk Terminology and common AI terms","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-risk-terminology/",0,"",""],["Turing's Red Flag","Toby Walsh","2015","paper","arXiv preprint","arxiv.org/abs/1510.09033",0,"","agents"],["Bad Universal Priors and Notions of Optimality","Jan Leike and Marcus Hutter","2015","paper","arXiv preprint","arxiv.org/abs/1510.04931",0,"","agents policy"],["A first look at the hard problem of corrigibility","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375041/a-first-look-at-the-hard-problem-of-corrigibility",0,"",""],["Asymptotic Logical Uncertainty and The Benford Test","Scott Garrabrant and 5 others","2015","paper","arXiv preprint","arxiv.org/abs/1510.03370",0,"",""],["Chatbots or set answers, not WBEs","Stuart_Armstrong","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf067037503c/chatbots-or-set-answers-not-wbes",0,"",""],["New report: “Leó Szilárd and the Danger of Nuclear Weapons”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/10/07/new-report-leo-szilard-and-the-danger-of-nuclear-weapons/",0,"",""],["Ambitious vs. narrow value learning","Paul Christiano","2015","report","medium.com","medium.com/ai-control/ambitious-vs-narrow-value-learning-99bd0c59847e",0,"",""],["October 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/10/03/october-2015-newsletter/",0,"",""],["New paper: “Asymptotic logical uncertainty and the Benford test”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/09/30/new-paper-asymptotic-logical-uncertainty-and-the-benford-test/",0,"",""],["Submission and Formatting Instructions for International Conference on Machine Learning (ICML 2015)","Your Name and Your CoAuthor’s Name","2015","paper","arXiv preprint","arxiv.org/abs/1509.08731",0,"","agents"],["The application of the secretary problem to real life dating","Elo","2015","blog","LessWrong","www.lesswrong.com/posts/Pk5Nyd5ByXwHWXX5r/the-application-of-the-secretary-problem-to-real-life-dating",0,"","theory"],["Quantilizers maximize expected utility subject to a conservative cost constraint","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375029/quantilizers-maximize-expected-utility-subject-to-a-conservative-cost-constraint",0,"",""],["Advisor games","Paul Christiano","2015","report","ai-alignment.com","ai-alignment.com/advisor-games-b33382fef68c",0,"",""],["Constructing Abstraction Hierarchies Using a Skill-Symbol Loop","George Konidaris","2015","paper","arXiv preprint","arxiv.org/abs/1509.07582",0,"","agents"],["Nomadism and Burning Man","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/09/20/nomadism-and-burning-man/",0,"",""],["Probabilities Small Enough To Ignore: An attack on Pascal's Mugging","Kaj_Sotala","2015","blog","LessWrong","www.lesswrong.com/posts/LxhJ8mhdBuX27BDug/probabilities-small-enough-to-ignore-an-attack-on-pascal-s",0,"","theory"],["September 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/09/14/september-2015-newsletter/",0,"",""],["How To Win The AI Box Experiment (Sometimes)","pinkgothic","2015","blog","LessWrong","www.lesswrong.com/posts/fbekxBfgvfc7pmnzB/how-to-win-the-ai-box-experiment-sometimes",0,"",""],["Maximal Maximum-Entropy Sets","jsteinhardt","2015","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2015/09/07/maximal-maximum-entropy-sets/",0,"",""],["Our summer fundraising drive is complete!","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/09/01/our-summer-fundraising-drive-is-complete/",0,"",""],["Confronting future catastrophic threats to humanity","Seth D. Baum and Bruce E. Tonn","2015","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0016328715001135",0,"",""],["The far future argument for confronting catastrophic threats to humanity: Practical significance and alternatives","Seth D. Baum","2015","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0016328715000312",0,"",""],["Final fundraiser day: Announcing our new team","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/08/31/final-fundraiser-day-announcing-our-new-team/",0,"",""],["Provability Counterfactuals vs Three Axioms of Galles and Pearl","IAFF-User-52","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670375017/provability-counterfactuals-vs-three-axioms-of-galles-and-pearl",0,"",""],["A Dialogue on Suffering Subroutines","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/a-dialogue-on-suffering-subroutines/",0,"",""],["A Lower Bound on the Importance of Promoting Cooperation","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/a-lower-bound-on-the-importance-of-promoting-cooperation/",0,"",""],["Differential Intellectual Progress as a Positive-Sum Project","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/differential-intellectual-progress-as-a-positive-sum-project/",0,"",""],["How Would Catastrophic Risks Affect Prospects for Compromise?","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/how-would-catastrophic-risks-affect-prospects-for-compromise/",0,"",""],["Reasons to Be Nice to Other Value Systems","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/reasons-to-be-nice-to-other-value-systems/",0,"",""],["AI and Effective Altruism","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/08/28/ai-and-effective-altruism/",0,"",""],["AI timelines and strategies","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-timelines-and-strategies/",0,"","forecasting"],["Posterior calibration and exploratory analysis for natural language processing models","Khanh Nguyen","2015","paper","arXiv preprint","arxiv.org/abs/1508.05154",0,"","evals"],["Powerful planners, not sentient software","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/08/18/powerful-planners-not-sentient-software/",0,"",""],["What Sets MIRI Apart?","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/08/14/what-sets-miri-apart/",0,"",""],["OOASP: Connecting Object-oriented and Logic Programming","Andreas Falkner and 3 others","2015","paper","arXiv preprint","arxiv.org/abs/1508.03032",0,"",""],["A response to Matthews on AI Risk","RyanCarey","2015","blog","EA Forum","forum.effectivealtruism.org/posts/a3PDjRBu9uTkRGeBS/a-response-to-matthews-on-ai-risk",0,"",""],["Assessing our past and potential impact","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/08/10/assessing-our-past-and-potential-impact/",0,"",""],["Target 3: Taking It To The Next Level","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/08/07/target-3-taking-it-to-the-next-level/",0,"",""],["AI Impacts research bounties","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-impacts-research-bounties/",0,"",""],["Introducing research bounties","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/introducing-research-bounties/",0,"",""],["The Technological Singularity","Murray Shanahan","2015","report","goodreads.com","www.goodreads.com/book/show/55605905-the-technological-singularity",0,"",""],["When AI Accelerates AI","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/08/03/when-ai-accelerates-ai/",0,"",""],["August 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/08/02/august-2015-newsletter/",0,"",""],["A Comprehensive Survey on Safe Reinforcement Learning","Javier Garcia and Fernando Fernandez","2015","report","jmlr.org","www.jmlr.org/papers/volume16/garcia15a/garcia15a.pdf",0,"",""],["A new MIRI FAQ, and other announcements","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/07/31/a-new-miri-faq-and-other-announcements/",0,"",""],["How to escape from your sandbox and from your hardware host","PhilGoetz","2015","blog","LessWrong","www.lesswrong.com/posts/TwH5jfkuvTatvAKEF/how-to-escape-from-your-sandbox-and-from-your-hardware-host",0,"",""],["Belief and Truth in Hypothesised Behaviours","Stefano V. Albrecht and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1507.07688",0,"","agents"],["MIRI’s Approach","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/27/miris-approach/",0,"",""],["Time flies when robots rule the earth","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/time-flies-when-robots-rule-the-earth/",0,"",""],["Brain performance in FLOPS","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/brain-performance-in-flops/",0,"",""],["Costs of human-level hardware","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/costs-of-human-level-hardware/",0,"",""],["Index of articles about hardware","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/index-of-hardware-articles/",0,"",""],["Systems I have tried: an overview","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/07/26/systems-i-have-tried-an-overview/",0,"",""],["Four Background Claims","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/24/four-background-claims/",0,"",""],["Cost of human-level information storage","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/cost-of-human-level-information-storage/",0,"",""],["Costs of information storage","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/costs-of-information-storage/",0,"",""],["Information storage in the brain","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/information-storage-in-the-brain/",0,"",""],["Asymptotic Logical Uncertainty: Concrete Failure of the Solomonoff Approach","Scott Garrabrant","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374fcb/asymptotic-logical-uncertainty-concrete-failure-of-the-solomonoff-approach",0,"","theory"],["Oracle AI: Human beliefs vs human values","Stuart_Armstrong","2015","blog","LessWrong","www.lesswrong.com/posts/b223mLTZNDFExf3Qp/oracle-ai-human-beliefs-vs-human-values",0,"",""],["Decision Maker based on Atomic Switches","Song-Ju Kim and 3 others","2015","paper","arXiv preprint","arxiv.org/abs/1507.05895",0,"",""],["An Idea For Corrigible, Recursively Improving Math Oracles","jimrandomh","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374fd3/an-idea-for-corrigible-recursively-improving-math-oracles",0,"",""],["Why Now Matters","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/20/why-now-matters/",0,"",""],["Targets 1 and 2: Growing MIRI","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/18/targets-1-and-2-growing-miri/",0,"",""],["An Astounding Year","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/16/an-astounding-year/",0,"",""],["MIRI’s 2015 Summer Fundraiser!","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/17/miris-2015-summer-fundraiser/",0,"",""],["Examples of AI's behaving badly","Stuart_Armstrong","2015","blog","LessWrong","www.lesswrong.com/posts/QDj5dozwPPe8aJ6ZZ/examples-of-ai-s-behaving-badly",0,"",""],["Event: Exercises in Economic Futurism","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/event-exercises-in-economic-futurism/",0,"",""],["Conversation with Steve Potter","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/conversation-with-steve-potter/",0,"",""],["Steve Potter on neuroscience and AI","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/steve-potter-on-neuroscience-and-ai/",0,"",""],["Inceptionism: Going deeper into neural networks","Alexander Mordvintsev and 2 others","2015","report","research.googleblog.com","research.googleblog.com/2015/06/inceptionism-going-deeper-into-neural.html",0,"",""],["Toward Idealized Decision Theory","Nate Soares and Benja Fallenstein","2015","paper","arXiv preprint","arxiv.org/abs/1507.01986",0,"","policy theory"],["Two big challenges in machine learning","Leon Bottou","2015","report","icml.cc","icml.cc/2015/invited/LeonBottouICML2015.pdf",0,"",""],["July 2015 Newsletter","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/07/05/july-2015-newsletter/",0,"",""],["New funding for AI Impacts","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/new-funding-for-ai-impacts/",0,"",""],["Vingean Reflection: Open Problems","abramdemski","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f9d/vingean-reflection-open-problems",0,"","agents"],["Grants and fundraisers","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/07/01/grants-fundraisers/",0,"",""],["Multitasking: Efﬁcient Optimal Planning for Bandit Superprocesses","Dylan Hadﬁeld-Menell and Stuart Russell","2015","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~russell/papers/uai15-multi.pdf",0,"",""],["New report: “The Asilomar Conference: A Case Study in Risk Mitigation”","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/06/30/new-report-the-asilomar-conference-a-case-study-in-risk-mitigation/",0,"",""],["Wanted: Office Manager (aka Force Multiplier)","Alex Vermeer","2015","blog","intelligence.org","intelligence.org/2015/07/01/wanted-office-manager/",0,"",""],["Two-boxing, smoking and chewing gum in Medical Newcomb problems","Caspar Oesterheld","2015","blog","LessWrong","www.lesswrong.com/posts/wWnN3y5GmqLLCJFAz/two-boxing-smoking-and-chewing-gum-in-medical-newcomb",0,"",""],["Recent developments in unifying logic and probability","Stuart Russell","2015","report","dl.acm.org","dl.acm.org/citation.cfm?doid=2797100.2699411",0,"",""],["Long-Term and Short-Term Challenges to Ensuring the Safety of AI Systems","jsteinhardt","2015","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2015/06/24/long-term-and-short-term-challenges-to-ensuring-the-safety-of-ai-systems/",0,"",""],["Sequential Extensions of Causal and Evidential Decision Theory","Tom Everitt and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1506.07359",0,"","agents theory"],["A simple model of the Löbstacle","orthonormal","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f8c/a-simple-model-of-the-loebstacle",0,"",""],["Agent Simulates Predictor using Second-Level Oracles","orthonormal","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f87/agent-simulates-predictor-using-second-level-oracles",0,"","agents"],["Dropout as a Bayesian Approximation: Representing Model Uncertainty in Deep Learning","Yarin Gal and Zoubin Ghahramani","2015","paper","arXiv preprint","arxiv.org/abs/1506.02142",0,"","robustness"],["Update on all the AI predictions","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/update-on-all-the-ai-predictions/",0,"",""],["Predictions of Human-Level AI Timelines","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/predictions-of-human-level-ai-timelines/",0,"","forecasting"],["Accuracy of AI Predictions","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/accuracy-of-ai-predictions/",0,"",""],["An Oracle standard trick","Stuart_Armstrong","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f85/an-oracle-standard-trick",0,"",""],["Introductions","Nate Soares","2015","blog","intelligence.org","intelligence.org/2015/05/31/introductions/",0,"",""],["June 2015 Newsletter","Jesse Galef","2015","blog","intelligence.org","intelligence.org/2015/06/01/june-2015-newsletter/",0,"",""],["Mortal universal agents & wireheading","Laurent Orseau","2015","report","www6.inrae.fr","www6.inrae.fr/mia-paris/Equipes/Membres/Anciens/Laurent-Orseau/Mortal-universal-agents-wireheading",0,"","reward-hacking agents"],["Publication biases toward shorter predictions","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/short-prediction-publication-biases/",0,"",""],["Selection bias from optimistic experts","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/bias-from-optimistic-predictors/",0,"",""],["Two papers accepted to AGI-15","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/05/29/two-papers-accepted-to-agi-15/",0,"",""],["Deep Learning","Yann LeCun and 2 others","2015","report","nature.com","www.nature.com/nature/journal/v521/n7553/abs/nature14539.html",0,"",""],["Probabilistic machine learning and artificial intelligence","Zoubin Ghahramani","2015","report","nature.com","www.nature.com/nature/journal/v521/n7553/full/nature14541.html",0,"",""],["Why do AGI researchers expect AI so soon?","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/why-do-agi-researchers-expect-ai-so-soon/",0,"",""],["Group Differences in AI Predictions","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/group-differences-in-ai-predictions/",0,"",""],["MIRI-related talks from the decision theory conference at Cambridge University","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/05/24/miri-related-talks-from-the-decision-theory-conference-at-cambridge-university/",0,"","theory"],["Supporting AI Impacts","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/supporting-ai-impacts/",0,"",""],["The Unreasonable Effectiveness of Recurrent Neural Networks","Andrej Karpathy","2015","report","karpathy.github.io","karpathy.github.io/2015/05/21/rnn-effectiveness/",0,"",""],["AI Timeline predictions in surveys and statements","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-timeline-predictions-in-surveys-and-statements/",0,"","forecasting"],["MIRI AI Predictions Dataset","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/miri-ai-predictions-dataset/",0,"",""],["The Maes-Garreau Law","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/the-maes-garreau-law/",0,"",""],["Weight Uncertainty in Neural Networks","Charles Blundell and 3 others","2015","paper","arXiv preprint","arxiv.org/abs/1505.05424",0,"",""],["Agents that can predict their Newcomb predictor","orthonormal","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f4f/agents-that-can-predict-their-newcomb-predictor",0,"","agents"],["What is Learning? A primary discussion about information and Representation","Hao Wu","2015","paper","arXiv preprint","arxiv.org/abs/1505.04813",0,"",""],["Hamming questions and bottlenecks","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/05/17/hamming-questions-and-bottlenecks/",0,"",""],["Optimal and Causal Counterfactual Worlds","Scott Garrabrant","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f4e/optimal-and-causal-counterfactual-worlds",0,"",""],["Automating change of representation for proofs in discrete mathematics","Daniel Raggi and 3 others","2015","paper","arXiv preprint","arxiv.org/abs/1505.02449",0,"",""],["A new approach to predicting brain-computer parity","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/tepsbrainestimate/",0,"",""],["Brain performance in TEPS","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/brain-performance-in-teps/",0,"",""],["A fond farewell and a new Executive Director","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/05/06/a-fond-farewell-and-a-new-executive-director/",0,"",""],["The Steering Problem","Paul Christiano","2015","report","ai-alignment.com","ai-alignment.com/the-steering-problem-a3543e65c5c4",0,"",""],["Metareasoning for Planning Under Uncertainty","Christopher H. Lin and 3 others","2015","paper","arXiv preprint","arxiv.org/abs/1505.00399",0,"","agents policy"],["May 2015 Newsletter","Jesse Galef","2015","blog","intelligence.org","intelligence.org/2015/05/01/may-2015-newsletter/",0,"",""],["New papers on reflective oracles and agents","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/04/28/new-papers-reflective/",0,"","agents"],["Glial Signaling","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/glial-signaling/",0,"",""],["Modal Bargaining Agents","orthonormal","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f20/modal-bargaining-agents",0,"","agents theory"],["Scale of the Human Brain","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/scale-of-the-human-brain/",0,"",""],["Concept Safety: Producing similar AI-human concept spaces","Kaj_Sotala","2015","blog","LessWrong","www.lesswrong.com/posts/q3N7hbhLjb6JCLEEg/concept-safety-producing-similar-ai-human-concept-spaces",0,"",""],["Neuron firing rates in humans","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/rate-of-neuron-firing/",0,"",""],["Towards Verifiably Ethical Robot Behaviour","Louise A. Dennis and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1504.03592",0,"","agents"],["Not so innocent: Reasoning about costs, competence, and culpability in very early childhood","Julian Jara-Ettinger and 2 others","2015","report","journals.sagepub.com","journals.sagepub.com/doi/10.1177/0956797615572806?url_ver=Z39.88-2003&rfr_id=ori:rid:crossref.org&rfr_dat=cr_pub%20%200pubmed",0,"","robustness"],["Technical and social approaches to AI safety","Paul Christiano","2015","report","medium.com","medium.com/ai-control/technical-and-social-approaches-to-ai-safety-5e225ca30c46",0,"",""],["Artificial Intelligence and Its Implications for Future Suffering","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/artificial-intelligence-and-its-implications-for-future-suffering/",0,"",""],["Flavors of Computation Are Flavors of Consciousness","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/flavors-of-computation-are-flavors-of-consciousness/",0,"",""],["Gains from Trade through Compromise","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/gains-from-trade-through-compromise/",0,"",""],["Metabolic Estimates of Rate of Cortical Firing","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/metabolic-estimates-of-rate-of-cortical-firing/",0,"",""],["International Cooperation vs. AI Arms Race","Brian Tomasik","2015","report","longtermrisk.org","longtermrisk.org/international-cooperation-vs-ai-arms-race/",0,"",""],["Preliminary prices for human-level hardware","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/preliminary-prices-for-human-level-hardware/",0,"",""],["Current FLOPS prices","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/current-flops-prices/",0,"",""],["Paraconsistent Tiling Agents (Very Early Draft)","IAFF-User-4","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f18/paraconsistent-tiling-agents-very-early-draft",0,"","agents"],["April 2015 newsletter","Jesse Galef","2015","blog","intelligence.org","intelligence.org/2015/04/01/april-2015-newsletter/",0,"",""],["Principles of Explanatory Debugging to Personalize Interactive Machine Learning","Todd Kulesza and 3 others","2015","report","dropline.net","dropline.net/wp-content/uploads/2012/02/iui-2015.pdf",0,"",""],["Superintelligence 29: Crunch time","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/H7kzai8uwPj9mQz9M/superintelligence-29-crunch-time",0,"","governance"],["Boxing an AI?","tailcalled","2015","blog","LessWrong","www.lesswrong.com/posts/z8s3bsw3WY9fdevSm/boxing-an-ai",0,"",""],["Negative visualization, radical acceptance and stoicism","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/03/26/negative-visualization-radical-acceptance-and-stoicism/",0,"",""],["Recent AI control brainstorming by Stuart Armstrong","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/27/recent-ai-control-brainstorming/",0,"","ai-control"],["Reflective oracles and the procrastination paradox","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f11/reflective-oracles-and-the-procrastination-paradox",0,"",""],["Shared Autonomy via Hindsight Optimization","Shervin Javdani and 2 others","2015","paper","arXiv preprint","arxiv.org/abs/1503.07619",0,"","agents"],["Superintelligence 28: Collaboration","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/uBzeBhySrQaoZkNCD/superintelligence-28-collaboration",0,"","governance"],["2014 in review","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/22/2014-review/",0,"",""],["Corrigible omniscient AI capable of making clones","Kaj_Sotala","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f03/corrigible-omniscient-ai-capable-of-making-clones",0,"",""],["Forum Digest: Reflective Oracles","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374f02/forum-digest-reflective-oracles",0,"",""],["The cost of TEPS","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/cost-of-teps/",0,"",""],["Introducing the Intelligent Agent Foundations Forum","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/18/introducing-intelligent-agent-foundations-forum/",0,"","agents theory"],["New report: “An Introduction to Löb’s Theorem in MIRI Research”","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/18/new-report-introduction-lobs-theorem-miri-research/",0,"",""],["Superintelligence 27: Pathways and enablers","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/sfGBkyDyu96eePZZ6/superintelligence-27-pathways-and-enablers",0,"","governance forecasting"],["Allen, The Singularity Isn’t Near","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/allen-the-singularity-isnt-near/",0,"",""],["Rationality: From AI to Zombies","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/03/12/rationality-ai-zombies/",0,"",""],["Kurzweil, The Singularity is Near","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/kurzweil-the-singularity-is-near/",0,"",""],["Minds: An Introduction","Rob Bensinger","2015","blog","LessWrong","www.lesswrong.com/posts/8GhSZzsQmusCN9is7/minds-an-introduction",0,"","theory"],["Wikipedia history of GFLOPS costs","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/wikipedia-history-of-gflops-costs/",0,"",""],["Bill Hibbard on Ethical Artificial Intelligence","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/09/bill-hibbard/",0,"",""],["Superintelligence 26: Science and technology strategy","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/kADkXCAq6aBBxSyqE/superintelligence-26-science-and-technology-strategy",0,"","governance"],["Trends in the cost of computing","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/trends-in-the-cost-of-computing/",0,"",""],["Fallenstein talk for APS March Meeting 2015","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/03/09/fallenstein-talk-aps-march-meeting-2015/",0,"",""],["Superintelligence 25: Components list for acquiring values","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/MFgj8hcTB9gjjL9rE/superintelligence-25-components-list-for-acquiring-values",0,"",""],["March 2015 newsletter","Jake","2015","blog","intelligence.org","intelligence.org/2015/03/01/march-newsletter-2/",0,"",""],["Ethical guidelines for a superintelligence","Ernest Davis","2015","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0004370214001453",0,"",""],["Sequential Feature Explanations for Anomaly Detection","Md Amran Siddiqui and Alan Fern and Thomas G. Dietterich and Weng-Keen Wong","2015","paper","arXiv preprint","arxiv.org/abs/1503.00038",0,"","evals benchmarks monitoring"],["What’s up with nuclear weapons?","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/whats-up-with-nuclear-weapons/",0,"",""],["Possible Empirical Investigations","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/possible-investigations/",0,"",""],["Human-level control through deep reinforcement learning","Volodymyr Mnih and 18 others","2015","report","nature.com","www.nature.com/nature/journal/v518/n7540/full/nature14236.html",0,"",""],["An Introduction to Löb’s Theorem in MIRI Research","Patrick LaVictoire","2015","report","intelligence.org","intelligence.org/files/lob-notes-IAFF.pdf",0,"",""],["From Seed AI to Technological Singularity via Recursively Self-Improving Software","Roman V. Yampolskiy","2015","paper","arXiv preprint","arxiv.org/abs/1502.06512",0,"",""],["Research topic: Hardware, software and AI","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/research-topic-hardware-software-and-ai/",0,"",""],["Oracle machines for automated philosophy","Nisan","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ed9/oracle-machines-for-automated-philosophy",0,"",""],["Superintelligence 23: Coherent extrapolated volition","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/EQFfj5eC5mqBMxF2s/superintelligence-23-coherent-extrapolated-volition",0,"",""],["Future of Life Institute’s recent milestones in AI safety","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/02/16/flis-recent-milestones-in-ai-safety/",0,"",""],["Un-manipulable counterfactuals","Stuart_Armstrong","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ecc/un-manipulable-counterfactuals",0,"",""],["An implementation of modal UDT","Benya_Fallenstein","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ed6/an-implementation-of-modal-udt",0,"","theory"],["Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift","Sergey Ioffe and Christian Szegedy","2015","paper","arXiv preprint","arxiv.org/abs/1502.03167",0,"",""],["How to study superintelligence strategy","Luke Muehlhauser","2015","report","lukemuehlhauser.com","lukemuehlhauser.com/some-studies-which-could-improve-our-strategic-picture-of-superintelligence/",0,"",""],["List of multipolar research projects","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/multipolar-research-projects/",0,"",""],["Multipolar research questions","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/multipolar-research-questions/",0,"",""],["Superintelligence 22: Emulation modulation and institutional design","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/NFTe38cwu7LqT2oTy/superintelligence-22-emulation-modulation-and-institutional",0,"","governance"],["How AI timelines are estimated","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/how-ai-timelines-are-estimated/",0,"","forecasting"],["UDT in the Land of Probabilistic Oracles","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ed2/udt-in-the-land-of-probabilistic-oracles",0,"",""],["At-least-human-level-at-human-cost AI","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/at-least-human-level-at-human-cost-ai/",0,"",""],["Davis on AI capability and motivation","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/02/06/davis-ai-capability-motivation/",0,"",""],["Non-manipulative oracles","Stuart_Armstrong","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374eca/non-manipulative-oracles",0,"",""],["Probabilistic Oracle Machines and Nash Equilibria","jessicata","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ec8/probabilistic-oracle-machines-and-nash-equilibria",0,"",""],["New annotated bibliography for MIRI’s technical agenda","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/02/05/new-annotated-bibliography-miris-technical-agenda/",0,"",""],["[LINK] Wait But Why - The AI Revolution Part 2","Adam Zerner","2015","blog","LessWrong","www.lesswrong.com/posts/ZqgCQQH6P6EPdCLaT/link-wait-but-why-the-ai-revolution-part-2",0,"",""],["From halting oracles to modal logic","Benya_Fallenstein","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374e6a/from-halting-oracles-to-modal-logic",0,"",""],["New mailing list for MIRI math/CS papers only","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/02/03/keep-date-miris-research-via-new-mailing-list/",0,"",""],["Superintelligence 21: Value learning","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/bFQfgwm72Zz9ZjTh4/superintelligence-21-value-learning",0,"",""],["Third-person counterfactuals","Benya_Fallenstein","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ebf/third-person-counterfactuals",0,"",""],["Discontinuous progress investigation","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/discontinuous-progress-investigation/",0,"",""],["February 2015 Newsletter","Jake","2015","blog","intelligence.org","intelligence.org/2015/02/01/february-2015-newsletter/",0,"",""],["Penicillin and syphilis","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/penicillin-and-syphilis/",0,"",""],["The odd counterfactuals of playing chicken","Benya_Fallenstein","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ec0/the-odd-counterfactuals-of-playing-chicken",0,"",""],["Vingean Reﬂection: Reliable Reasoning for Self-Improving Agents","Benja Fallenstein and Nate Soares","2015","report","intelligence.org","intelligence.org/files/VingeanReflection.pdf",0,"","agents"],["New report: “The value learning problem”","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/29/new-report-value-learning-problem/",0,"",""],["Superintelligence 20: The value-loading problem","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/FP8T6rdZ3ohXxJRto/superintelligence-20-the-value-loading-problem",0,"",""],["The AI Revolution: Our Immortality or Extinction","Tim Urban","2015","report","waitbutwhy.com","waitbutwhy.com/2015/01/artificial-intelligence-revolution-2.html",0,"",""],["Multibit reflective oracles","Benya_Fallenstein","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ebe/multibit-reflective-oracles",0,"",""],["Benefits and Risks of Artificial Intelligence","Thomas G. Dietterich","2015","report","medium.com","medium.com/@tdietterich/benefits-and-risks-of-artificial-intelligence-460d288cccf3",0,"",""],["Benefits and Risks of Artificial Intelligence","Thomas G. Dietterich","2015","report","medium.com","medium.com/@tdietterich/benefits-and-risks-of-artificial-intelligence-460d288cccf3#.4mobx01nw",0,"",""],["New, Brief Popular-Level Introduction to AI Risks and Superintelligence","LyleN","2015","blog","LessWrong","www.lesswrong.com/posts/p3QMGoKPdHtCWgPBD/new-brief-popular-level-introduction-to-ai-risks-and",0,"",""],["An Introduction to Löb's Theorem in MIRI Research","orthonormal","2015","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374ebb/an-introduction-to-loeb-s-theorem-in-miri-research",0,"",""],["Formalizing Two Problems of Realistic World Models","So8res","2015","blog","LessWrong","www.lesswrong.com/posts/uyMhKJCNCcfPtEFLj/formalizing-two-problems-of-realistic-world-models",0,"","theory"],["List of Analyses of Time to Human-Level AI","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/list-of-analyses-of-time-to-human-level-ai/",0,"",""],["New report: “Formalizing Two Problems of Realistic World Models”","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/22/new-report-formalizing-two-problems-realistic-world-models/",0,"",""],["The AI Revolution: The Road to Superintelligence","Tim Urban","2015","report","waitbutwhy.com","waitbutwhy.com/2015/01/artificial-intelligence-revolution-1.html",0,"",""],["The slow traversal of ‘human-level’","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/the-slow-traversal-of-human-level/",0,"",""],["Making or breaking a thinking machine","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/making-or-breaking-a-thinking-machine/",0,"",""],["The range of human intelligence","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/is-the-range-of-human-intelligence-small/",0,"",""],["Are AI surveys seeing the inside view?","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/are-ai-surveys-seeing-the-inside-view/",0,"",""],["Visualizing Representations: Deep Learning and Human Beings","Chris Olah","2015","report","colah.github.io","colah.github.io/posts/2015-01-Visualizing-Representations/.",0,"",""],["New report: “Vingean Reflection: Reliable Reasoning for Self-Improving Agents”","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/15/new-report-vingean-reflection-reliable-reasoning-self-improving-agents/",0,"","agents"],["Event: Multipolar AI workshop with Robin Hanson","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/event-multipolar-ai-workshop-with-robin-hanson/",0,"",""],["Selfish preferences and self-modification","Manfred","2015","blog","LessWrong","www.lesswrong.com/posts/zgbZNwW7f3C89ZgGK/selfish-preferences-and-self-modification",0,"","theory"],["Michie and overoptimism","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/michie-and-overoptimism/",0,"",""],["Superintelligence 18: Life in an algorithmic economy","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/iZNcMkS6ghqBQA24E/superintelligence-18-life-in-an-algorithmic-economy",0,"",""],["Were nuclear weapons cost-effective explosives?","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/were-nuclear-weapons-cost-effective-explosives/",0,"",""],["2014-15 New Year review","Victoria Krakovna","2015","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2015/01/11/2014-15-new-year-review/",0,"",""],["An improved “AI Impacts” website","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/11/improved-ai-impacts-website/",0,"",""],["A summary of AI surveys","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/a-summary-of-ai-surveys/",0,"",""],["AI Timeline Surveys","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-timeline-surveys/",0,"","forecasting"],["Michie Survey","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/michie-survey/",0,"",""],["AI and the Big Nuclear Discontinuity","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/ai-and-the-big-nuclear-discontinuity/",0,"",""],["New report: “Questions of reasoning under logical uncertainty”","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/09/new-report-questions-reasoning-logical-uncertainty/",0,"",""],["The AI Impacts Blog","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/the-ai-impacts-blog/",0,"",""],["The Biggest Technological Leaps","Katja Grace","2015","blog","aiimpacts.org","aiimpacts.org/the-biggest-technological-leaps/",0,"",""],["Brooks and Searle on AI volition and timelines","Rob Bensinger","2015","blog","intelligence.org","intelligence.org/2015/01/08/brooks-searle-agi-volition-timelines/",0,"","forecasting"],["Matthias Troyer on Quantum Computers","Luke Muehlhauser","2015","blog","intelligence.org","intelligence.org/2015/01/07/matthias-troyer-quantum-computers/",0,"",""],["Superintelligence 17: Multipolar scenarios","KatjaGrace","2015","blog","LessWrong","www.lesswrong.com/posts/8QgNrNPaoyZeEY4ZD/superintelligence-17-multipolar-scenarios",0,"","governance"],["January 2015 Newsletter","Jake","2015","blog","intelligence.org","intelligence.org/2015/01/01/january-2015-newsletter/",0,"",""],["Bayesian computational models for inferring preferences","Owain Rhys Evans","2015","report","dspace.mit.edu","dspace.mit.edu/handle/1721.1/101522",0,"",""],["Corrigibility","Nate Soares and 3 others","2015","report","intelligence.org","intelligence.org/files/Corrigibility.pdf",0,"",""],["Death and pain of a digital brain","Anders Sandberg","2015","report","doi.org","doi.org/10.1016/S0262-4079(15)31174-X",0,"",""],["Deep learning in neural networks: An overview","Jürgen Schmidhuber","2015","report","sciencedirect.com","www.sciencedirect.com/science/article/pii/S0893608014002135",0,"",""],["Existential risk and existential hope: definitions","Owen Cotton-Barratt and Toby Ord","2015","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/Existential-risk-and-existential-hope.pdf",0,"",""],["Global challenges: 12 risks that threaten human civilization","Dennis Pamlin and Stuart Armstrong","2015","report","ncbi.nlm.nih.gov","www.ncbi.nlm.nih.gov/pmc/articles/PMC7228299/",0,"",""],["How valuable is movement growth?","Owen Cotton-Barratt","2015","report","globalprioritiesproject.org","globalprioritiesproject.org/wp-content/uploads/2015/05/MovementGrowth.pdf",0,"",""],["How We’re Predicting AI – or Failing to","Stuart Armstrong and Kaj Sotala","2015","report","link.springer.com","link.springer.com/10.1007/978-3-319-09668-1_2",0,"",""],["How we’re predicting AI–or failing to","Stuart Armstrong and Kaj Sotala","2015","report","intelligence.org","intelligence.org/files/PredictingAI.pdf",0,"",""],["Learning the Preferences of Bounded Agents","Owain Evans and Noah D Goodman","2015","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/nips-workshop-2015-website.pdf",0,"","agents"],["Mathematics of Machine Learning","Philippe Rigollet","2015","report","ocw.mit.edu","ocw.mit.edu/courses/mathematics/18-657-mathematics-of-machine-learning-fall-2015/lecture-notes/MIT18_657F15_LecNote.pdf",0,"",""],["Moral Trade","Toby Ord","2015","report","amirrorclear.net","www.amirrorclear.net/files/moral-trade.pdf",0,"",""],["Motivated value selection for artificial agents","Stuart Armstrong","2015","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/2015/03/Armstrong_AAAI_2015_Motivated_Value_Selection.pdf",0,"","agents"],["Outrunning the Law: Extraterrestrial Liberty and Universal Colonisation","Stuart Armstrong and 2 others","2015","report","link.springer.com","link.springer.com/chapter/10.1007/978-3-319-09567-7_11",0,"",""],["Oﬀ-policy Monte Carlo agents with variable behaviour policies","Stuart Armstrong","2015","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/monte_carlo_arXiv.pdf",0,"","agents policy"],["Proof-Producing Reflection for HOL","Benja Fallenstein and Ramana Kumar","2015","report","link.springer.com","link.springer.com/chapter/10.1007/978-3-319-22102-1_11",0,"",""],["Research priorities for robust and beneficial artificial intelligence: an open letter","Stuart Russell and 2 others","2015","report","futureoflife.org","futureoflife.org/open-letter/ai-open-letter/",0,"",""],["Responses to catastrophic AGI risk: a survey","Kaj Sotala and Roman V Yampolskiy","2015","report","iopscience.iop.org","iopscience.iop.org/article/10.1088/0031-8949/90/1/018001",0,"",""],["Treating anthropic selfish preferences as an extension of TDT","Manfred","2015","blog","LessWrong","www.lesswrong.com/posts/gTmWZEu3CcEQ6fLLM/treating-anthropic-selfish-preferences-as-an-extension-of",0,"","theory"],["We, Borg: Speculations on hive minds as a posthuman state","Anders Sandberg","2015","report","aleph.se","www.aleph.se/Trans/Global/Posthumanity/WeBorg.html",0,"",""],["Cases of Discontinuous Technological Progress","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/cases-of-discontinuous-technological-progress/",0,"",""],["Effect of nuclear weapons on historic trends in explosives","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/discontinuity-from-nuclear-weapons/",0,"",""],["Superintelligence 16: Tool AIs","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/sL8hCYecDwcrRhfCT/superintelligence-16-tool-ais",0,"",""],["AGI-09 Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/agi-09-survey/",0,"",""],["AI@50 Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/ai50-survey/",0,"",""],["Bainbridge Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/bainbridge-survey/",0,"",""],["Early Views of AI","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/early-views-of-ai/",0,"",""],["FHI Winter Intelligence Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/fhi-ai-timelines-survey/",0,"",""],["Hanson AI Expert Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/hanson-ai-expert-survey/",0,"",""],["Klein AGI Survey","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/klein-agi-survey/",0,"",""],["Kruel AI Interviews","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/kruel-ai-survey/",0,"",""],["Müller and Bostrom AI Progress Poll","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/muller-and-bostrom-ai-progress-poll/",0,"",""],["Similarity Between Historical and Contemporary AI Predictions","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/similarity-between-historical-and-contemporary-ai-predictions/",0,"",""],["Open and closed mental states","Victoria Krakovna","2014","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2014/12/26/open-and-closed-mental-states/",0,"",""],["Our new technical research agenda overview","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/23/new-technical-research-agenda-overview/",0,"",""],["Superintelligence 15: Oracles, genies and sovereigns","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/yTy2Fp8Wm7m8rHHz5/superintelligence-15-oracles-genies-and-sovereigns",0,"",""],["Adam: A Method for Stochastic Optimization","Diederik P. Kingma and Jimmy Ba","2014","paper","arXiv preprint","arxiv.org/abs/1412.6980",0,"",""],["Explaining and Harnessing Adversarial Examples","Ian J. Goodfellow and Jonathon Shlens & Christian Szegedy","2014","paper","arXiv preprint","arxiv.org/abs/1412.6572",0,"","deception robustness"],["2014 Winter Matching Challenge Completed!","Malo Bourgon","2014","blog","intelligence.org","intelligence.org/2014/12/18/2014-winter-matching-challenge-completed/",0,"",""],["New report: “Computable probability distributions which converge…”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/16/new-report-computable-probability-distributions-converge/",0,"",""],["New report: “Toward Idealized Decision Theory”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/16/new-report-toward-idealized-decision-theory/",0,"","theory"],["New report: “Tiling agents in causal graphs”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/16/new-report-tiling-agents-causal-graphs/",0,"","agents"],["Superintelligence 14: Motivation selection methods","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/FBEaheqfmDgL6gB5x/superintelligence-14-motivation-selection-methods",0,"",""],["Superintelligence 13: Capability control methods","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/398Swu6jmczzSRvHy/superintelligence-13-capability-control-methods",0,"","agents"],["Concept Learning for Safe Autonomous AI","Kaj Sotala","2014","report","intelligence.org","intelligence.org/files/ConceptLearning.pdf",0,"",""],["Deep Neural Networks are Easily Fooled: High Confidence Predictions for Unrecognizable Images","Anh Nguyen and 2 others","2014","paper","arXiv preprint","arxiv.org/abs/1412.1897",0,"",""],["New paper: “Concept learning for safe autonomous AI”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/05/new-paper-concept-learning-safe-autonomous-ai/",0,"",""],["2014 Winter Matching Challenge!","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/01/2014-winter-matching-challenge/",0,"",""],["December newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/12/01/december-newsletter/",0,"",""],["Superintelligence 12: Malignant failure modes","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/BqoE5vhPNCB7X6Say/superintelligence-12-malignant-failure-modes",0,"","goodharts-law"],["The great downside dilemma for risky emerging technologies","Seth D Baum","2014","report","stacks.iop.org","stacks.iop.org/1402-4896/89/i=12/a=128004?key=crossref.f5938bc78a3023d740968f020cfa9970",0,"",""],["Three impacts of machine intelligence","Paul Christiano","2014","report","medium.com","medium.com/@paulfchristiano/three-impacts-of-machine-intelligence-6285c8d85376",0,"",""],["Superintelligence 11: The treacherous turn","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/B39GNTsN3HocW8KFo/superintelligence-11-the-treacherous-turn",0,"","deception"],["xkcd on the AI box experiment","FiftyTwo","2014","blog","LessWrong","www.lesswrong.com/posts/4uY7pciyfmMFgWDis/xkcd-on-the-ai-box-experiment",0,"",""],["Superintelligence 10: Instrumentally convergent goals","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/BD6G9wzRRt3fxckNC/superintelligence-10-instrumentally-convergent-goals",0,"","instrumental-convergence"],["Three misconceptions in Edge.org’s conversation on “The Myth of AI”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/11/18/misconceptions-edge-orgs-conversation-myth-ai/",0,"",""],["Musk on AGI Timeframes","Artaxerxes","2014","blog","LessWrong","www.lesswrong.com/posts/kzHJ5BRhgkj9CSQ3N/musk-on-agi-timeframes",0,"","forecasting"],["Stable self-improvement as a research problem","paulfchristiano","2014","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374e92/stable-self-improvement-as-a-research-problem",0,"",""],["Simplicity priors with reflective oracles","Benya_Fallenstein","2014","blog","AI Alignment Forum","www.alignmentforum.org/posts/5bd75cc58225bf0670374e76/simplicity-priors-with-reflective-oracles",0,"",""],["Logical Limitations to Machine Ethics with Consequences to Lethal Autonomous Weapons","Matthias Englert and 2 others","2014","paper","arXiv preprint","arxiv.org/abs/1411.2842",0,"","agents"],["Superintelligence 9: The orthogonality of intelligence and goals","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/FtAJZWCMps7FWKTT3/superintelligence-9-the-orthogonality-of-intelligence-and",0,"",""],["Video of Bostrom’s talk on Superintelligence at UC Berkeley","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/11/06/video-bostroms-talk-superintelligence-uc-berkeley/",0,"",""],["A new guide to MIRI’s research","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/11/06/new-guide-miris-research/",0,"",""],["Ethical Artificial Intelligence","Bill Hibbard","2014","paper","arXiv preprint","arxiv.org/abs/1411.1373",0,"","evals instrumental-convergence agents"],["MIRI’s November Newsletter","Jake","2014","blog","intelligence.org","intelligence.org/2014/11/01/miris-november-newsletter/",0,"",""],["The Financial Times story on MIRI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/31/financial-times-story-miri/",0,"",""],["Do Artificial Reinforcement-Learning Agents Matter Morally?","Brian Tomasik","2014","paper","arXiv preprint","arxiv.org/abs/1410.8233",0,"","agents"],["New report: “UDT with known search order”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/30/new-report-udt-known-search-order/",0,"",""],["Singularity2014.com appears to be a fake","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/27/singularity2014-fake/",0,"",""],["Superintelligence 7: Decisive strategic advantage","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/vkjWGJrFWBnzHtxrw/superintelligence-7-decisive-strategic-advantage",0,"","governance"],["Superintelligence 6: Intelligence explosion kinetics","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/GT8uvxBjidrmM3MCv/superintelligence-6-intelligence-explosion-kinetics",0,"","forecasting"],["Introducing Corrigibility (an FAI research subfield)","So8res","2014","blog","LessWrong","www.lesswrong.com/posts/yFr8ZfGGnRX5GqndZ/introducing-corrigibility-an-fai-research-subfield",0,"",""],["New paper: “Corrigibility”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/18/new-report-corrigibility/",0,"",""],["Importance motivation: a double-edged sword","Victoria Krakovna","2014","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2014/10/18/importance-motivation-a-double-edged-sword/",0,"",""],["The Precautionary Principle (with Application to the Genetic Modification of Organisms)","Nassim Nicholas Taleb1 and 4 others","2014","paper","arXiv preprint","arxiv.org/abs/1410.5787",0,"",""],["AGI outcomes and civilizational competence","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/16/agi-outcomes-civilizational-competence/",0,"",""],["Superintelligence 5: Forms of Superintelligence","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/semvkn56ZFcXBNc2d/superintelligence-5-forms-of-superintelligence",0,"",""],["Nate Soares’ talk: “Why ain’t you rich?”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/07/nate-soares-talk-aint-rich/",0,"",""],["MIRI’s October Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/10/01/october-newsletter/",0,"",""],["Superintelligence Reading Group 3: AI and Uploads","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/hopzyM5ckzMNHwcQR/superintelligence-reading-group-3-ai-and-uploads",0,"",""],["Newcomblike problems are the norm","So8res","2014","blog","LessWrong","www.lesswrong.com/posts/puutBJLWbg2sXpFbu/newcomblike-problems-are-the-norm",0,"","theory"],["Superintelligence Reading Group 2: Forecasting AI","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/56b8n8FT6fksnDZwY/superintelligence-reading-group-2-forecasting-ai",0,"","forecasting"],["CEV-tropes","snarles","2014","blog","LessWrong","www.lesswrong.com/posts/6bFBkk3XNiTwgE8R3/cev-tropes",0,"",""],["CEV: coherence versus extrapolation","Stuart_Armstrong","2014","blog","LessWrong","www.lesswrong.com/posts/RGd85AErgmXmAMKw5/cev-coherence-versus-extrapolation",0,"",""],["Superintelligence Reading Group - Section 1: Past Developments and Present Capabilities","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/mmZ2PaRo86pDXu8ii/superintelligence-reading-group-section-1-past-developments",0,"",""],["Kristinn Thórisson on constructivist AI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/14/kris-thorisson/",0,"",""],["Nate Soares speaking at Purdue University","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/12/nate-soares-speaking-purdue-september-18th/",0,"",""],["Good policy ideas that won’t happen (yet)","Niel_Bowerman","2014","blog","EA Forum","forum.effectivealtruism.org/posts/n5CNeo9jxDsCit9dj/good-policy-ideas-that-won-t-happen-yet",0,"","governance policy robustness"],["Ken Hayworth on brain emulation prospects","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/09/hayworth/",0,"",""],["Sequence to Sequence Learning with Neural Networks","Ilya Sutskever and 2 others","2014","paper","arXiv preprint","arxiv.org/abs/1409.3215",0,"",""],["Friendly AI Research Help from MIRI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/08/friendly-ai-research-help-miri/",0,"",""],["Citadel house sessions – a year in review","Victoria Krakovna","2014","blog","vkrakovna.wordpress.com","vkrakovna.wordpress.com/2014/09/07/citadel-house-sessions-a-year-in-review/",0,"",""],["Daniel Roy on probabilistic programming and AI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/04/daniel-roy/",0,"",""],["Goal retention discussion with Eliezer","MaxTegmark","2014","blog","LessWrong","www.lesswrong.com/posts/FtNFhuXXtmSjnNvE7/goal-retention-discussion-with-eliezer",0,"","instrumental-convergence"],["John Fox on AI safety","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/04/john-fox/",0,"",""],["Very Deep Convolutional Networks for Large-Scale Image Recognition","Karen Simonyan and Andrew Zisserman","2014","paper","arXiv preprint","arxiv.org/abs/1409.1556",0,"","evals"],["Friendly Artificial Intelligence: the Physics Challenge","Max Tegmark","2014","paper","In proceedings of the AAAI 2015 Workshop On AI and Ethics, p87,\n  Toby Walsh, Ed. (2015)","arxiv.org/abs/1409.0813",0,"",""],["MIRI’s September Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/09/01/september-newsletter-2/",0,"",""],["Neural Machine Translation by Jointly Learning to Align and Translate","Dzmitry Bahdanau and 2 others","2014","paper","arXiv preprint","arxiv.org/abs/1409.0473",0,"",""],["Superintelligence reading group","Katja Grace","2014","blog","intelligence.org","intelligence.org/2014/08/31/superintelligence-reading-group/",0,"",""],["Superintelligence reading group","KatjaGrace","2014","blog","LessWrong","www.lesswrong.com/posts/QDmzDZ9CEHrKQdvcn/superintelligence-reading-group",0,"",""],["Knightian uncertainty: a rejection of the MMEU rule","So8res","2014","blog","LessWrong","www.lesswrong.com/posts/SEov2u5Y7mJTPaTLK/knightian-uncertainty-a-rejection-of-the-mmeu-rule",0,"","theory"],["New paper: “Exploratory engineering in artificial intelligence”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/08/22/new-paper-exploratory-engineering-artificial-intelligence/",0,"",""],["2014 Summer Matching Challenge Completed!","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/08/15/2014-summer-matching-challenge-completed/",0,"",""],["MIRI’s recent effective altruism talks","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/08/11/miris-recent-effective-altruism-talks/",0,"",""],["A Logic for Reasoning about Upper Probabilities","Joseph Y. Halpern and Riccardo Pucella","2014","paper","arXiv preprint","arxiv.org/abs/1408.1485",0,"",""],["Groundwork for AGI safety engineering","Rob Bensinger","2014","blog","intelligence.org","intelligence.org/2014/08/04/groundwork-ai-safety-engineering/",0,"",""],["MIRI’s August 2014 newsletter","Jake","2014","blog","intelligence.org","intelligence.org/2014/08/01/august-newsletter-2/",0,"",""],["Algorithmic Game Theory (Fall 2013)","Tim Roughgarden","2014","report","timroughgarden.org","timroughgarden.org/f13/f13.pdf",0,"",""],["Scott Frickel on intellectual movements","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/07/28/scott-frickel/",0,"",""],["Evidence with Uncertain Likelihoods","Joseph Y. Halpern and Riccardo Pucella","2014","paper","arXiv preprint","arxiv.org/abs/1407.7189",0,"","agents robustness"],["Beyond Point-and-Shoot Morality: Why Cognitive (Neuro)Science Matters for Ethics","Joshua D Greene","2014","report","static1.squarespace.com","static1.squarespace.com/static/54763f79e4b0c4e55ffb000c/t/54e90604e4b09706d4a4fc65/1424557572437/beyond-point-and-shoot-morality.pdf",0,"",""],["Nick Bostrom to speak about Superintelligence at UC Berkeley","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/07/25/bostrom/",0,"",""],["A Visualization of Nick Bostrom’s Superintelligence","anonymous","2014","blog","LessWrong","www.lesswrong.com/posts/ukmDvowTpe2NboAsX/a-visualization-of-nick-bostrom-s-superintelligence",0,"",""],["2014 Summer Matching Challenge!","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/07/21/2014-summer-matching-challenge/",0,"",""],["Knightian Uncertainty and Ambiguity Aversion: Motivation","So8res","2014","blog","LessWrong","www.lesswrong.com/posts/iDYaCJ3o3Q7ypriTF/knightian-uncertainty-and-ambiguity-aversion-motivation",0,"","theory"],["Representing and Reasoning about Game Strategies","Dongmo Zhang and Michael Thielsher","2014","paper","arXiv preprint","arxiv.org/abs/1407.5380",0,"","robustness"],["May 2015 decision theory conference at Cambridge University","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/07/12/may-2015-decision-theory-workshop-cambridge/",0,"","theory"],["How to study superintelligence strategy","Luke Muehlhauser","2014","report","lukemuehlhauser.com","lukemuehlhauser.com/some-studies-which-could-improve-our-strategic-picture-of-superintelligence/",0,"",""],["Superintelligence: Paths, Dangers, Strategies","Nick Bostrom","2014","report","goodreads.com","www.goodreads.com/book/show/20527133-superintelligence",0,"",""],["The errors, insights and lessons of famous AI predictions – and what they mean for the future","Stuart Armstrong and 2 others","2014","report","tandfonline.com","www.tandfonline.com/doi/full/10.1080/0952813X.2014.895105",0,"",""],["MIRI’s July 2014 newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/07/01/july-newsletter-2/",0,"",""],["New report: “Non-omniscience, probabilistic inference, and metamathematics”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/23/new-report-non-omniscience-probabilistic-inference-metamathematics/",0,"",""],["Roger Schell on long-term computer security research","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/23/roger-schell/",0,"",""],["New chapter in Cambridge Handbook of Artificial Intelligence","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/19/new-chapter-cambridge-handbook-artificial-intelligence/",0,"",""],["Our mid-2014 strategic plan","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/11/mid-2014-strategic-plan/",0,"",""],["Generative Adversarial Networks","Ian J. Goodfellow and 7 others","2014","paper","arXiv preprint","arxiv.org/abs/1406.2661",0,"","evals training-data"],["Allan Friedman on cybersecurity and cyberwar","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/06/allan-friedman-cybersecurity-cyberwar/",0,"",""],["New report: “Distributions allowing tiling of staged subjective EU maximizers”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/06/new-report-distributions-allowing-tiling-staged-subjective-eu-maximizers/",0,"",""],["MIRI’s June 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/06/01/miris-june-2014-newsletter/",0,"",""],["Dropout: A Simple Way to Prevent Neural Networks from Overfitting","Nitish Srivastava and 4 others","2014","report","jmlr.org","jmlr.org/papers/volume15/srivastava14a/srivastava14a.pdf",0,"",""],["Milind Tambe on game theory in security applications","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/30/milind-tambe/",0,"",""],["New report: “Loudness: On priors over preference relations”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/30/new-report-loudness-priors-preference-relations/",0,"",""],["Aaron Tomb on crowd-sourced formal verification","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/29/aaron-tomb/",0,"",""],["MIRI wants to fund your independently-organized Friendly AI workshop","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/29/miri-wants-fund-independently-organized-friendly-ai-workshop/",0,"",""],["Lennart Beringer on the Verified Software Toolchain","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/27/lennart-beringer/",0,"",""],["Decision Theory with Resource-Bounded Agents","Joseph Y. Halpern and 2 others","2014","report","doi.wiley.com","doi.wiley.com/10.1111/tops.12088",0,"","agents theory"],["Johann Schumann on high-assurance systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/24/johann-schumann/",0,"","assurance"],["Sandor Veres on autonomous agents","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/23/sandor-veres/",0,"","agents"],["New Paper: “Program Equilibrium in the Prisoner’s Dilemma via Löb’s Theorem”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/17/new-paper-program-equilibrium-prisoners-dilemma-via-lobs-theorem/",0,"",""],["Paul Christiano – Machine intelligence and capital accumulation","Tessa","2014","blog","EA Forum","forum.effectivealtruism.org/posts/DjECTMZy9jB5hGZwg/paul-christiano-machine-intelligence-and-capital",0,"",""],["Christof Koch and Stuart Russell on machine superintelligence","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/13/christof-koch-stuart-russell-machine-superintelligence/",0,"",""],["Exponential and non-exponential trends in information technology","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/12/exponential-and-non-exponential/",0,"",""],["Benjamin Pierce on clean-slate security architectures","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/11/benjamin-pierce/",0,"",""],["Michael Fisher on verifying autonomous systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/09/michael-fisher/",0,"","agents"],["Harry Buhrman on quantum algorithms and cryptography","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/07/harry-buhrman/",0,"",""],["Liveblogging the SV Gives Fundraiser","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/06/liveblogging-the-svgives-fundraiser/",0,"",""],["New paper: “Problems of self-reference in self-improving space-time embedded intelligence”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/06/new-paper-problems-of-self-reference-in-self-improving-space-time-embedded-intelligence/",0,"",""],["Calling all MIRI supporters for unique giving opportunity!","Malo Bourgon","2014","blog","intelligence.org","intelligence.org/2014/05/04/calling-all-miri-supporters/",0,"",""],["Kasper Stoy on self-reconfigurable robots","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/02/kasper-stoy/",0,"",""],["MIRI’s May 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/05/01/miris-may-2014-newsletter/",0,"",""],["New Paper: “The errors, insights, and lessons of famous AI predictions”","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/30/new-paper-the-errors-insights-and-lessons-of-famous-ai-predictions/",0,"",""],["Suresh Jagannathan on higher-order program verification","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/30/suresh-jagannathan-on-higher-order-program-verification/",0,"",""],["Ruediger Schack on quantum Bayesianism","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/29/ruediger-schack/",0,"",""],["David J. Atkinson on autonomous systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/28/david-j-atkinson/",0,"","agents"],["Help MIRI in a Massive 24-Hour Fundraiser on May 6th","Louie Helm","2014","blog","intelligence.org","intelligence.org/2014/04/25/may-6th-miri-participating-in-massive-24-hour-online-fundraiser/",0,"",""],["Domitilla del Vecchio on hybrid control for autonomous vehicles","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/24/domitilla-del-vecchio/",0,"",""],["Roland Siegwart on autonomous mobile robots","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/25/roland-siegwart/",0,"",""],["Ariel Procaccia on economics and computation","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/23/ariel-procaccia/",0,"",""],["Dave Doty on algorithmic self-assembly","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/23/dave-doty/",0,"",""],["Martin Hilbert on the world’s information capacity","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/22/martin-hilbert/",0,"",""],["Suzana Herculano-Houzel on cognitive ability and brain size","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/22/suzana-herculano-houzel/",0,"",""],["Why MIRI?","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/20/why-miri/",0,"",""],["AI risk, new executive summary","Stuart_Armstrong","2014","blog","LessWrong","www.lesswrong.com/posts/BZfnJGe5S6KtB5pjQ/ai-risk-new-executive-summary",0,"",""],["Thomas Bolander on self-reference and agent introspection","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/13/thomas-bolander/",0,"","agents"],["Botworld: a cellular automaton for studying self-modifying agents embedded in their environment","So8res","2014","blog","LessWrong","www.lesswrong.com/posts/Ai4zMKQTX86fMtHN3/botworld-a-cellular-automaton-for-studying-self-modifying",0,"","agents theory"],["Jonathan Millen on covert channel communication","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/12/jonathan-millen/",0,"",""],["MIRI’s April 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/10/miris-april-2014-newsletter/",0,"",""],["New Report: Botworld","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/10/new-report-botworld/",0,"",""],["Wolf Kohn on hybrid systems control","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/11/wolf-kohn/",0,"",""],["Limitations and risks of machine ethics","Miles Brundage","2014","report","tandfonline.com","www.tandfonline.com/doi/abs/10.1080/0952813X.2014.895108",0,"",""],["Diana Spears on the safety of adaptive agents","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/09/diana-spears/",0,"","agents"],["Paulo Tabuada on program synthesis for cyber-physical systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/09/paulo-tabuada/",0,"",""],["Will MacAskill on normative uncertainty","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/08/will-macaskill/",0,"",""],["Siren worlds and the perils of over-optimised search","Stuart_Armstrong","2014","blog","AI Alignment Forum","www.alignmentforum.org/posts/nFv2buafNc9jSaxAH/siren-worlds-and-the-perils-of-over-optimised-search",0,"",""],["Enabling Automatic Certification of Online Auctions","Wei Bai and 2 others","2014","paper","EPTCS 147, 2014, pp. 123-132","arxiv.org/abs/1404.0854",0,"","agents assurance robustness"],["Erik DeBenedictis on supercomputing","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/03/erik-debenedictis/",0,"",""],["2013 in Review: Fundraising","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/04/02/2013-in-review-fundraising/",0,"",""],["Anil Nerode on hybrid systems control","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/26/anil-nerode/",0,"",""],["Lyle Ungar on forecasting","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/26/lyle-ungar/",0,"","forecasting"],["Michael Carbin on integrity properties in approximate computing","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/23/michael-carbin/",0,"",""],["Randal Koene on whole brain emulation","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/20/randal-a-koene-on-whole-brain-emulation/",0,"",""],["Max Tegmark on the mathematical universe","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/19/max-tegmark/",0,"",""],["MIRI’s March 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/18/miris-march-2014-newsletter/",0,"",""],["Recent Hires at MIRI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/13/hires/",0,"",""],["Toby Walsh on computational social choice","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/10/toby-walsh/",0,"",""],["Randall Larsen and Lynne Kidder on USA bio-response","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/09/randall-larsen-and-lynne-kidder/",0,"",""],["John Ridgway on safety-critical systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/08/john-ridgway-on-safety-critical-systems/",0,"",""],["David Cook on the VV&A process","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/07/david-cook/",0,"",""],["How to Study Unsafe AGI's safely (and why we might have no choice)","Punoxysm","2014","blog","LessWrong","www.lesswrong.com/posts/eJDTaBEZgCpbSdAbS/how-to-study-unsafe-agi-s-safely-and-why-we-might-have-no",0,"",""],["Robert Constable on correct-by-construction programming","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/02/bob-constable/",0,"",""],["Anders Sandberg on Space Colonization","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/02/anders-sandberg/",0,"",""],["Armando Tacchella on Safety in Future AI Systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/03/02/armando-tacchella/",0,"",""],["The world’s distribution of computation (initial findings)","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/28/the-worlds-distribution-of-computation-initial-findings/",0,"",""],["Nik Weaver on Paradoxes of Rational Agency","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/24/nik-weaver-on-paradoxes-of-rational-agency/",0,"",""],["MIRI’s May 2014 Workshop","Alex Vermeer","2014","blog","intelligence.org","intelligence.org/2014/02/22/miris-may-2014-workshop/",0,"",""],["SUDT: A toy decision theory for updateless anthropics","Benya","2014","blog","AI Alignment Forum","www.alignmentforum.org/posts/NaZPjaLPCGZWdTyrL/sudt-a-toy-decision-theory-for-updateless-anthropics",0,"","theory"],["Conversation with Holden Karnofsky about Future-Oriented Philanthropy","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/21/conversation-with-holden-karnofsky-about-future-oriented-philanthropy/",0,"",""],["John Baez on Research Tactics","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/21/john-baez-on-research-tactics/",0,"",""],["2013 in Review: Friendly AI Research","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/18/2013-in-friendly-ai-research/",0,"",""],["MIRI’s February 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/17/miris-february-2014-newsletter/",0,"",""],["New eBook: ‘Smarter Than Us’","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/17/new-ebook-smarter-than-us/",0,"",""],["André Platzer on Verifying Cyber-Physical Systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/15/andre-platzer-on-verifying-cyber-physical-systems/",0,"",""],["Conversation with Jacob Steinhardt about MIRI Strategy","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/11/conversation-with-jacob-steinhardt-about-miri-strategy/",0,"",""],["Gerwin Klein on Formal Methods","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/11/gerwin-klein-on-formal-methods/",0,"",""],["A Fervent Defense of Frequentist Statistics","jsteinhardt","2014","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2014/02/10/a-fervent-defense-of-frequentist-statistics/",0,"",""],["2013 in Review: Strategic and Expository Research","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/08/2013-in-review-strategic-and-expository-research/",0,"",""],["L-zombies! (L-zombies?)","Benya","2014","blog","LessWrong","www.lesswrong.com/posts/7nAxgQYGYrEY5ZCAD/l-zombies-l-zombies",0,"","theory"],["MIRI’s Experience with Google Adwords","Louie Helm","2014","blog","intelligence.org","intelligence.org/2014/02/06/miris-experience-with-google-adwords/",0,"",""],["Careers at MIRI","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/03/careers-at-miri/",0,"",""],["Ronald de Wolf on Quantum Computing","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/03/ronald-de-wolf-on-quantum-computing/",0,"",""],["Robust Cooperation: A Case Study in Friendly AI Research","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/02/01/robust-cooperation-a-case-study-in-friendly-ai-research/",0,"",""],["Smarter than us: The rise of machine intelligence","Stuart Armstrong","2014","report","goodreads.com","www.goodreads.com/book/show/20830144-smarter-than-us",0,"",""],["Emil Vassev on Formal Verification","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/30/emil-vassev-on-formal-verification/",0,"",""],["Mike Frank on reversible computing","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/31/mike-frank-on-reversible-computing/",0,"",""],["Two MIRI talks from AGI-11","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/31/two-miri-talks-from-agi-11/",0,"",""],["Existential Risk Strategy Conversation with Holden Karnofsky","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/27/existential-risk-strategy-conversation-with-holden-karnofsky/",0,"",""],["How Big is the Field of Artificial Intelligence? (initial findings)","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/28/how-big-is-ai/",0,"",""],["Google may be trying to take over the world","anonymous","2014","blog","LessWrong","www.lesswrong.com/posts/YgHQ9Nez4S8277Wr2/google-may-be-trying-to-take-over-the-world",0,"","governance"],["Human-Level AI","Katja Grace","2014","blog","aiimpacts.org","aiimpacts.org/human-level-ai/",0,"",""],["Robust Cooperation in the Prisoner’s Dilemma: Program Equilibrium via Provability Logic","Mihaly Barasz and 2 others","2014","paper","arXiv preprint","arxiv.org/abs/1401.5577",0,"","agents"],["2013 in Review: Outreach","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/20/2013-in-review-outreach/",0,"",""],["The Second Machine Age: Work, Progress, and Prosperity in a Time of Brilliant Technologies","Erik Brynjolfsson and Andrew McAfee","2014","report","goodreads.com","www.goodreads.com/book/show/23316526-the-second-machine-age",0,"",""],["Exploiting Model Equivalences for Solving Interactive Dynamic Influence Diagrams","Yifeng Zeng and Prashant Doshi","2014","paper","Journal Of Artificial Intelligence Research, Volume 43, pages\n  211-255, 2012","arxiv.org/abs/1401.4600",0,"","agents policy"],["Want to help MIRI by investing in XRP?","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/18/investing-in-xrp/",0,"",""],["MIRI’s January 2014 Newsletter","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/17/miris-january-2014-newsletter/",0,"",""],["Reasoning About the Transfer of Control","Wiebe van der Hoek and 2 others","2014","paper","Journal Of Artificial Intelligence Research, Volume 37, pages\n  437-477, 2010","arxiv.org/abs/1401.3825",0,"","deception agents"],["Networks of Influence Diagrams: A Formalism for Representing Agents' Beliefs and Decision-Making Processes","Yaakov Gal and Avi Pfeffer","2014","paper","Journal Of Artificial Intelligence Research, Volume 33, pages\n  109-147, 2008","arxiv.org/abs/1401.3426",0,"","agents"],["MIRI strategy conversation with Steinhardt, Karnofsky, and Amodei","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/13/miri-strategy-conversation-with-steinhardt-karnofsky-and-amodei/",0,"",""],["Kathleen Fisher on High-Assurance Systems","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/10/kathleen-fisher-on-high-assurance-systems/",0,"","assurance"],["Donor Story #1: Noticing Inferential Distance","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2014/01/05/donor-story-1-giving-after-critique/",0,"",""],["Another Critique of Effective Altruism","jsteinhardt","2014","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2014/01/05/another-critique-of-effective-altruism/",0,"",""],["7 New Technical Reports, and a New Paper","Luke Muehlhauser","2014","blog","intelligence.org","intelligence.org/2013/12/31/7-new-technical-reports-and-a-new-paper/",0,"",""],["Active reward learning","Christian Daniel et al","2014","report","roboticsproceedings.org","www.roboticsproceedings.org/rss10/p31.pdf",0,"","policy"],["Being nice to software animals and babies","Anders Sandberg","2014","report","doi.org","doi.org/10.1002/9781118736302.ch20",0,"",""],["Ethics of brain emulations","Anders Sandberg","2014","report","researchgate.net","www.researchgate.net/publication/263474325_Ethics_of_Brain_Emulations",0,"",""],["Explaining data-driven document classifications","David Martens and Foster Provost","2014","report","pages.stern.nyu.edu","pages.stern.nyu.edu/~fprovost/Papers/martens-CeDER-11-01.pdf",0,"",""],["Exploratory Engineering in AI","Luke Muehlhauser and Bill Hibbard","2014","report","ssec.wisc.edu","www.ssec.wisc.edu/~billh/g/CACM2014.pdf",0,"",""],["Hail Mary, value porosity, and utility diversification","Nick Bostrom","2014","report","nickbostrom.com","nickbostrom.com/papers/porosity.pdf",0,"",""],["Integrating Human Observer Inferences into Robot Motion Planning","Anca Dragan and Siddhartha Srinivasa","2014","report","ri.cmu.edu","www.ri.cmu.edu/publications/integrating-human-observer-inferences-into-robot-motion-planning/",0,"",""],["Introduction—The Transhumanist FAQ: A General Introduction","Nick Bostrom","2014","report","nickbostrom.com","nickbostrom.com/views/transhumanist.pdf",0,"",""],["Justification Narratives for Individual Classifications","Or Biran and Kathleen McKeown","2014","report","cs.columbia.edu","www.cs.columbia.edu/~orb/papers/justification_automl_2014.pdf",0,"",""],["Monte Carlo model of brain emulation development","Anders Sandberg","2014","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/reports/2014-1.pdf",0,"",""],["Problems of Self-reference in Self-improving Space-Time Embedded Intelligence","Benja Fallenstein and Nate Soares","2014","report","link.springer.com","link.springer.com/10.1007/978-3-319-09274-4_3",0,"",""],["Reinforcement Learning and the Reward Engineering Principle","Daniel Dewey","2014","report","danieldewey.net","www.danieldewey.net/reward-engineering-principle.pdf",0,"",""],["Safe Exploration Techniques for Reinforcement Learning – An Overview","Martin Pecka and Tomas Svoboda","2014","report","cmp.felk.cvut.cz","cmp.felk.cvut.cz/~peckama2/papers/safe_exploration_overview_lncs.pdf",0,"",""],["The ethics of artificial intelligence","Nick Bostrom and Eliezer Yudkowsky","2014","report","cambridge.org","www.cambridge.org/core/product/identifier/CBO9781139046855A027/type/book_part",0,"",""],["The five biggest threats to human existence","Anders Sandberg","2014","report","theconversation.com","theconversation.com/the-five-biggest-threats-to-human-existence-27053",0,"",""],["Transhumanism and the Meaning of Life","Anders Sandberg","2014","report","aleph.se","www.aleph.se/papers/Meaning%20of%20life.pdf",0,"",""],["Unifying Logic and Probability: A New Dawn for AI?","Stuart Russell","2014","report","link.springer.com","link.springer.com/10.1007/978-3-319-08795-5_2",0,"",""],["Unprecedented technological risks","Nick Beckstead and 6 others","2014","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Unprecedented-Technological-Risks.pdf",0,"",""],["Visualizing and Understanding Convolutional Networks","Matthew D Zeiler and Rob Fergus","2014","report","cs.nyu.edu","www.cs.nyu.edu/~fergus/papers/zeilerECCV2014.pdf",0,"",""],["Who knows anything about anything about AI?","Stuart Armstrong and Seán ÓhÉigeartaigh","2014","report","doi.org","doi.org/10.1002/9781118736302.ch3",0,"",""],["Why we need friendly AI","Luke Muehlhauser and Nick Bostrom","2014","report","nickbostrom.com","nickbostrom.com/views/whyfriendlyai.pdf",0,"",""],["Convex Conditions for Strong Convexity","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/12/30/linfty-strong-convexity/",0,"",""],["Winter 2013 Fundraiser Completed!","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/26/winter-2013-fundraiser-completed/",0,"",""],["Building Phenomenological Bridges","Rob Bensinger","2013","report","lesswrong.com","lesswrong.com/lw/jd9/building_phenomenological_bridges/",0,"",""],["Intriguing properties of neural networks","Christian Szegedy","2013","paper","arXiv preprint","arxiv.org/abs/1312.6199",0,"","interpretability deception"],["Josef Urban on Machine Learning and Automated Reasoning","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/21/josef-urban-on-machine-learning-and-automated-reasoning/",0,"",""],["2013 in Review: Operations","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/20/2013-in-review-operations/",0,"",""],["Auto-Encoding Variational Bayes","Diederik P Kingma and Max Welling","2013","paper","arXiv preprint","arxiv.org/abs/1312.6114",0,"",""],["Deep Inside Convolutional Networks: Visualising Image Classification Models and Saliency Maps","Karen Simonyan","2013","paper","arXiv preprint","arxiv.org/abs/1312.6034",0,"",""],["Giving the AI definition a form suitable for the engineer","Dimiter Dobrev","2013","paper","arXiv preprint","arxiv.org/abs/1312.5713",0,"",""],["Playing Atari with Deep Reinforcement Learning","Volodymyr Mnih and 6 others","2013","paper","arXiv preprint","arxiv.org/abs/1312.5602",0,"",""],["New Paper: “Why We Need Friendly AI”","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/18/new-paper-why-we-need-friendly-ai/",0,"",""],["MIRI’s December 2013 Newsletter","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/16/miris-december-2013-newsletter/",0,"",""],["Scott Aaronson on Philosophical Progress","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/13/aaronson/",0,"",""],["International cooperation vs. AI arms race","Brian_Tomasik","2013","blog","LessWrong","www.lesswrong.com/posts/vw9QAviBxcGodMHfN/international-cooperation-vs-ai-arms-race",0,"","governance"],["2013 Winter Matching Challenge","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/02/2013-winter-matching-challenge/",0,"",""],["New Paper: “Predicting AGI: What can we say when we know so little?”","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/12/01/new-paper-predicting-agi-what-can-we-say-when-we-know-so-little/",0,"",""],["Knowing Whether","Jie Fan and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1312.0144",0,"",""],["New Paper: “Racing to the Precipice”","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/11/27/new-paper-racing-to-the-precipice/",0,"",""],["MIRI’s November 2013 Newsletter","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/11/18/miri-update-november-2013/",0,"",""],["Quantum versus logical bombs","Stuart_Armstrong","2013","blog","LessWrong","www.lesswrong.com/posts/JGHQPybvjLAgimXae/quantum-versus-logical-bombs",0,"","theory"],["Visualizing and Understanding Convolutional Networks","Matthew D Zeiler and Rob Fergus","2013","paper","arXiv preprint","arxiv.org/abs/1311.2901",0,"","benchmarks"],["Support MIRI by Shopping at AmazonSmile","Alex Vermeer","2013","blog","intelligence.org","intelligence.org/2013/11/06/amazonsmile/",0,"",""],["Greg Morrisett on Secure and Reliable Systems","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/11/05/greg-morrisett-on-secure-and-reliable-systems-2/",0,"",""],["From Philosophy to Math to Engineering","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/11/04/from-philosophy-to-math-to-engineering/",0,"",""],["Kidnapping and the game of Chicken","Manfred","2013","blog","LessWrong","www.lesswrong.com/posts/GQy2BSQG9Dd6vPhs8/kidnapping-and-the-game-of-chicken",0,"","theory"],["Lone Genius Bias and Returns on Additional Researchers","ChrisHallquist","2013","blog","LessWrong","www.lesswrong.com/posts/qZHn4rDBXdkPKnwNH/lone-genius-bias-and-returns-on-additional-researchers",0,"","forecasting"],["Robin Hanson on Serious Futurism","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/11/01/robin-hanson/",0,"",""],["New Paper: “Embryo Selection for Cognitive Enhancement”","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/30/new-paper-embryo-selection-for-cognitive-enhancement/",0,"",""],["Markus Schmidt on Risks from Novel Biotechnologies","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/28/markus-schmidt-on-risks-from-novel-biotechnologies/",0,"",""],["Bas Steunebrink on Self-Reflective Programming","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/25/bas-steunebrink-on-sleight/",0,"",""],["Game Theory with Translucent Players","Joseph Y. Halpern and Rafael Pass","2013","paper","arXiv preprint","arxiv.org/abs/1310.6438",0,"",""],["Probabilistic Metamathematics and the Definability of Truth","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/23/probabilistic-metamathematics-and-the-definability-of-truth/",0,"",""],["Hadi Esmaeilzadeh on Dark Silicon","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/21/hadi-esmaeilzadeh-on-dark-silicon/",0,"",""],["Russell and Norvig on Friendly AI","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/19/russell-and-norvig-on-friendly-ai/",0,"",""],["Ben Goertzel on AGI as a Field","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/18/ben-goertzel/",0,"",""],["Richard Posner on AI Dangers","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/18/richard-posner-on-ai-dangers/",0,"",""],["Distributed Representations of Words and Phrases and their Compositionality","Tomas Mikolov and 4 others","2013","paper","arXiv preprint","arxiv.org/abs/1310.4546",0,"","robustness"],["MIRI’s October Newsletter","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/12/miris-october-newsletter/",0,"",""],["Empowerment -- an Introduction","Christoph Salge and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1310.1863",0,"","agents"],["The Relevance of Proofs of the Rationality of Probability Theory to Automated Reasoning and Cognitive Models","Ernest Davis","2013","paper","arXiv preprint","arxiv.org/abs/1310.1328",0,"",""],["Mathematical Proofs Improve But Don’t Guarantee Security, Safety, and Friendliness","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/03/proofs/",0,"",""],["Upcoming Talks at Harvard and MIT","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/10/01/upcoming-talks-at-harvard-and-mit/",0,"",""],["I played the AI Box Experiment again! (and lost both games)","Tuxedage","2013","blog","LessWrong","www.lesswrong.com/posts/oexwJBd3zAjw9Cru8/i-played-the-ai-box-experiment-again-and-lost-both-games",0,"",""],["Paul Rosenbloom on Cognitive Architectures","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/25/paul-rosenbloom-interview/",0,"",""],["Probability, knowledge, and meta-probability","David_Chapman","2013","blog","LessWrong","www.lesswrong.com/posts/2xmKZu73gZLDEQw7c/probability-knowledge-and-meta-probability",0,"","theory"],["Double Your Donations via Corporate Matching","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/14/double-your-donation/",0,"",""],["Effective Altruism and Flow-Through Effects","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/14/effective-altruism-and-flow-through-effects/",0,"",""],["How well will policy-makers handle AGI? (initial findings)","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/12/how-well-will-policy-makers-handle-agi-initial-findings/",0,"","policy"],["MIRI’s September Newsletter","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/10/september-newsletter/",0,"",""],["The Ultimate Newcomb's Problem","Eliezer Yudkowsky","2013","blog","LessWrong","www.lesswrong.com/posts/RAh4fekdiRhZxb2Kw/the-ultimate-newcomb-s-problem",0,"","theory"],["Laurent Orseau on Artificial General Intelligence","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/06/laurent-orseau-on-agi/",0,"",""],["Five Theses, Using Only Simple Words","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/05/five-theses-using-only-simple-words/",0,"",""],["I attempted the AI Box Experiment again! (And won - Twice!)","Tuxedage","2013","blog","LessWrong","www.lesswrong.com/posts/dop3rLwFhW5gtpEgz/i-attempted-the-ai-box-experiment-again-and-won-twice",0,"",""],["How effectively can we plan for future decades? (initial findings)","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/09/04/how-effectively-can-we-plan-for-future-decades/",0,"",""],["The Hanson-Yudkowsky AI-Foom Debate is now available as an eBook!","Alex Vermeer","2013","blog","intelligence.org","intelligence.org/2013/09/04/the-hanson-yudkowsky-ai-foom-debate-is-now-available-as-an-ebook/",0,"",""],["Stephen Hsu on Cognitive Genomics","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/31/stephen-hsu-on-cognitive-genomics/",0,"",""],["MIRI’s November 2013 Workshop in Oxford","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/30/miris-november-2013-workshop-in-oxford/",0,"",""],["Holden Karnofsky on Transparent Research Analyses","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/25/holden-karnofsky-interview/",0,"",""],["Transparency in Safety-Critical Systems","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/25/transparency-in-safety-critical-systems/",0,"","interpretability"],["Three Impacts of Machine Intelligence","Paul_Christiano","2013","blog","EA Forum","forum.effectivealtruism.org/posts/KdxGwxwY3t7iw9xjB/three-impacts-of-machine-intelligence",0,"","governance"],["2013 Summer Matching Challenge Completed!","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/21/2013-summer-matching-challenge-completed/",0,"",""],["Formalization, Mechanization and Automation of Gödel's Proof of God's Existence","Christoph Benzmüller and Bruno Woltzenlogel Paleo","2013","paper","Frontiers in Artificial Intelligence and Applications, Volume 263:\n  ECAI 2014","arxiv.org/abs/1308.4526",0,"",""],["Game Theory with Translucent Players","Joseph Y. Halpern and Rafael Pass","2013","paper","arXiv preprint","arxiv.org/abs/1308.3778",0,"",""],["Luke at Quixey on Tuesday (Aug. 20th)","staff","2013","blog","intelligence.org","intelligence.org/2013/08/16/luke-at-quixey-on-tuesday-aug-20th/",0,"",""],["August Newsletter: New Research and Expert Interviews","Jake","2013","blog","intelligence.org","intelligence.org/2013/08/13/august-newsletter/",0,"",""],["What is AGI?","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/11/what-is-agi/",0,"",""],["How to Measure Anything","lukeprog","2013","blog","LessWrong","www.lesswrong.com/posts/ybYBCK9D7MZCcdArB/how-to-measure-anything",0,"","theory"],["Benja Fallenstein on the Löbian Obstacle to Self-Modifying Systems","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/04/benja-interview/",0,"",""],["“Algorithmic Progress in Six Domains” Released","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/08/02/algorithmic-progress-in-six-domains-released/",0,"",""],["Action, Outcome, and Value: A Dual System Framework for Morality","Fiery Cushman","2013","report","journals.sagepub.com","journals.sagepub.com/doi/10.1177/1088868313495594?url_ver=Z39.88-2003&rfr_id=ori:rid:crossref.org&rfr_dat=cr_pub%20%200pubmed",0,"",""],["AI Risk and the Security Mindset","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/31/ai-risk-and-the-security-mindset/",0,"",""],["Man and Machine: Questions of Risk, Trust and Accountability in Today's AI Technology","Piyush Ahuja","2013","paper","arXiv preprint","arxiv.org/abs/1307.7127",0,"","interpretability"],["Index of Transcripts","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/25/index-of-transcripts/",0,"",""],["MIRI’s December 2013 Workshop","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/24/miris-december-2013-workshop/",0,"",""],["Nick Beckstead on the Importance of the Far Future","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/17/beckstead-interview/",0,"",""],["Roman Yampolskiy on AI Safety Engineering","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/15/roman-interview/",0,"",""],["James Miller on Unusual Incentives Facing AGI Companies","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/12/james-miller-interview/",0,"",""],["MIRI’s July Newsletter: Fundraiser and New Papers","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/11/july-newsletter/",0,"",""],["2013 Summer Matching Challenge!","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/08/2013-summer-matching-challenge/",0,"",""],["A Knowledge-based Treatment of Human-Automation Systems","Yoram Moses and Marcia K. Shamo","2013","paper","arXiv preprint","arxiv.org/abs/1307.2191",0,"","agents"],["Evidential Decision Theory, Selection Bias, and Reference Classes","Qiaochu_Yuan","2013","blog","LessWrong","www.lesswrong.com/posts/fWKGXSZ3uXxLKAxvm/evidential-decision-theory-selection-bias-and-reference",0,"","theory"],["MIRI Has Moved!","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/08/miri-has-moved/",0,"",""],["MIRI’s September 2013 Workshop","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/08/miris-september-2013-workshop/",0,"",""],["Responses to Catastrophic AGI Risk: A Survey","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/07/08/responses-to-catastrophic-agi-risk-a-survey/",0,"",""],["For FAI: Is \"Molecular Nanotechnology\" putting our best foot forward?","leplen","2013","blog","LessWrong","www.lesswrong.com/posts/9JKdnAakjCtvxTReJ/for-fai-is-molecular-nanotechnology-putting-our-best-foot",0,"","forecasting"],["What is Intelligence?","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/06/19/what-is-intelligence-2/",0,"",""],["Why do theists, undergrads, and Less Wrongers favor one-boxing on Newcomb?","CarlShulman","2013","blog","LessWrong","www.lesswrong.com/posts/5vmRXMxFLvSe2a9CM/why-do-theists-undergrads-and-less-wrongers-favor-one-boxing",0,"",""],["Convexity counterexample","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/06/12/convexity-counterexample/",0,"",""],["Do Earths with slower economic growth have a better chance at FAI?","Eliezer Yudkowsky","2013","blog","LessWrong","www.lesswrong.com/posts/FS6NCWzzP8DHp4aD4/do-earths-with-slower-economic-growth-have-a-better-chance",0,"","forecasting"],["MIRI’s July 2013 Workshop","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/06/07/miris-july-2013-workshop/",0,"",""],["New Research Page and Two New Articles","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/06/06/new-research-page-and-two-new-articles/",0,"",""],["Robust Cooperation in the Prisoner's Dilemma","orthonormal","2013","blog","LessWrong","www.lesswrong.com/posts/iQWk5jYeDg5ACCmpx/robust-cooperation-in-the-prisoner-s-dilemma",0,"","theory"],["Friendly AI Research as Effective Altruism","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/06/05/friendly-ai-research-as-effective-altruism/",0,"",""],["Mahatma Armstrong: CEVed to death.","Stuart_Armstrong","2013","blog","LessWrong","www.lesswrong.com/posts/vgFvnr7FefZ3s3tHp/mahatma-armstrong-ceved-to-death",0,"",""],["MIRI May Newsletter: Intelligence Explosion Microeconomics and Other Publications","Jake","2013","blog","intelligence.org","intelligence.org/2013/05/30/miri-may-newsletter-intelligence-explosion-microeconomics-and-other-publications/",0,"",""],["New Transcript: Yudkowsky and Aaronson","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/29/new-transcript-yudkowsky-and-aaronson/",0,"",""],["Sign up for DAGGRE to improve science & technology forecasting","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/24/sign-up-for-daggre-to-improve-science-technology-forecasting/",0,"","forecasting"],["Four Articles Added to Research Page","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/24/four-articles-added-to-research-page/",0,"",""],["When Will AI Be Created?","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/15/when-will-ai-be-created/",0,"",""],["Advise MIRI with Your Domain-Specific Expertise","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/15/advise-miri-with-your-domain-specific-expertise/",0,"",""],["Pascal's Muggle: Infinitesimal Priors and Strong Evidence","Eliezer Yudkowsky","2013","blog","LessWrong","www.lesswrong.com/posts/Ap4KfkHyxjYPDiqh2/pascal-s-muggle-infinitesimal-priors-and-strong-evidence",0,"","theory"],["Five theses, two lemmas, and a couple of strategic implications","Eliezer Yudkowsky","2013","blog","intelligence.org","intelligence.org/2013/05/05/five-theses-two-lemmas-and-a-couple-of-strategic-implications/",0,"",""],["AGI Impact Experts and Friendly AI Experts","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/05/01/agi-impacts-experts-and-friendly-ai-experts/",0,"",""],["New report: Intelligence Explosion Microeconomics","Eliezer Yudkowsky","2013","blog","LessWrong","www.lesswrong.com/posts/CZQuFoqgPXQawH9aL/new-report-intelligence-explosion-microeconomics",0,"","forecasting"],["“Intelligence Explosion Microeconomics” Released","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/29/intelligence-explosion-microeconomics-released/",0,"",""],["“Singularity Hypotheses” Published","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/25/singularity-hypotheses-published/",0,"",""],["Altair’s Timeless Decision Theory Paper Published","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/19/altairs-timeless-decision-theory-paper-published/",0,"","theory"],["Interactive POMDP Lite: Towards Practical Planning to Predict and Exploit Intentions for Interacting with Self-Interested Agents","Trong Nghia Hoang and Kian Hsiang Low","2013","paper","arXiv preprint","arxiv.org/abs/1304.5159",0,"","evals agents"],["MIRI’s April newsletter: Relaunch Celebration and a New Math Result","Jake","2013","blog","intelligence.org","intelligence.org/2013/04/18/miri-april-newsletter-relaunch-celebration-and-a-new-math-result/",0,"",""],["Facing the Intelligence Explosion ebook","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/13/facing-the-intelligence-explosion-ebook/",0,"",""],["MIRI’s Strategy for 2013","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/13/miris-strategy-for-2013/",0,"",""],["Formal Verification of Distributed Aircraft Controllers","Sarah M. Loos and 2 others","2013","report","symbolaris.com","symbolaris.com/pub/discworld.pdf",0,"",""],["The Lean Nonprofit","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/04/04/the-lean-nonprofit/",0,"",""],["A Backwards View for Assessment","Ross D. Shachter and David Heckerman","2013","paper","arXiv preprint","arxiv.org/abs/1304.3107",0,"",""],["A Difficulty in the Concept of CEV","anonymous","2013","blog","LessWrong","www.lesswrong.com/posts/6pBPiEGqS8ncNq8x4/a-difficulty-in-the-concept-of-cev",0,"",""],["An Empirical Comparison of Three Inference Methods","David Heckerman","2013","paper","arXiv preprint","arxiv.org/abs/1304.2357",0,"","evals robustness"],["Dempster-Shafer vs. Probabilistic Logic","Daniel Hunter","2013","paper","arXiv preprint","arxiv.org/abs/1304.2713",0,"",""],["Generating Decision Structures and Causal Explanations for Decision Making","Spencer Star","2013","paper","arXiv preprint","arxiv.org/abs/1304.2376",0,"","agents"],["Inference Policies","Paul E. Lehner","2013","paper","arXiv preprint","arxiv.org/abs/1304.1516",0,"","policy"],["Integrating Logical and Probabilistic Reasoning for Decision Making","John S. Breese and Edison Tse","2013","paper","arXiv preprint","arxiv.org/abs/1304.2751",0,"",""],["Reasoning About Beliefs and Actions Under Computational Resource Constraints","Eric J. Horvitz","2013","paper","arXiv preprint","arxiv.org/abs/1304.2759",0,"","deception theory"],["When Should a Decision Maker Ignore the Advice of a Decision Aid?","Paul E. Lehner and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1304.1515",0,"",""],["Why AI may not foom","John_Maxwell","2013","blog","LessWrong","www.lesswrong.com/posts/77xLbXs6vYQuhT8hq/why-ai-may-not-foom",0,"","forecasting"],["Early draft of naturalistic reflection paper","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/03/22/early-draft-of-naturalistic-reflection-paper/",0,"",""],["An Approximate Nonmyopic Computation for Value of Information","David Heckerman and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1303.5720",0,"","deception"],["Probability Estimation in Face of Irrelevant Information","Adam J. Grove and Daphne Koller","2013","paper","arXiv preprint","arxiv.org/abs/1303.5719",0,"","deception agents theory"],["AI prediction case study 5: Omohundro's AI drives","Stuart_Armstrong","2013","blog","LessWrong","www.lesswrong.com/posts/gbKhdCLNrAebarXNM/ai-prediction-case-study-5-omohundro-s-ai-drives",0,"","instrumental-convergence"],["Probabilistic Abstractions I","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/03/15/probabilistic-abstractions-i/",0,"",""],["Pairwise Independence vs. Independence","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/03/13/pairwise-independence-vs-independence/",0,"",""],["A problem with \"playing chicken with the universe\" as an approach to UDT","Karl","2013","blog","LessWrong","www.lesswrong.com/posts/9KoyMKHmwCCJdMma4/a-problem-with-playing-chicken-with-the-universe-as-an",0,"","theory"],["March Newsletter","Jake","2013","blog","intelligence.org","intelligence.org/2013/03/07/march-newsletter/",0,"",""],["Upcoming MIRI Research Workshops","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/03/07/upcoming-miri-research-workshops/",0,"",""],["A Synthesis of Logical and Probabilistic Reasoning for Program Understanding and Debugging","Lisa J. Burnell and Eric J. Horvitz","2013","paper","arXiv preprint","arxiv.org/abs/1303.1488",0,"","robustness"],["Constructing Lower Probabilities","Carl G. Wagner and Bruce Tonn","2013","paper","arXiv preprint","arxiv.org/abs/1303.1516",0,"",""],["Tradeoffs in Constructing and Evaluating Temporal Influence Diagrams","Gregory M. Provan","2013","paper","arXiv preprint","arxiv.org/abs/1303.1458",0,"","evals"],["Two Procedures for Compiling Influence Diagrams","Paul E. Lehner and Azar Sadigh","2013","paper","arXiv preprint","arxiv.org/abs/1303.1494",0,"",""],["Meta Decision Theory and Newcomb's Problem","wdmacaskill","2013","blog","LessWrong","www.lesswrong.com/posts/CttZFMmikuKuFmNT7/meta-decision-theory-and-newcomb-s-problem",0,"","theory"],["Decision Theory FAQ","lukeprog","2013","blog","LessWrong","www.lesswrong.com/posts/zEWJBFFMvQ835nq6h/decision-theory-faq",0,"","theory"],["Welcome to Intelligence.org","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/02/28/welcome-to-intelligence-org/",0,"",""],["Anytime Decision Making with Imprecise Probabilities","Michael Pittarelli","2013","paper","arXiv preprint","arxiv.org/abs/1302.6837",0,"",""],["Exploiting the Rule Structure for Decision Making within the Independent Choice Logic","David L. Poole","2013","paper","arXiv preprint","arxiv.org/abs/1302.4978",0,"","evals agents"],["Is There a Role for Qualitative Risk Assessment?","Paul J. Krause and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1302.4970",0,"","robustness"],["Independence with Lower and Upper Probabilities","Lonnie Chrisman","2013","paper","arXiv preprint","arxiv.org/abs/1302.3568",0,"",""],["A Fun Optimization Problem","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/02/09/a-fun-optimization-problem/",0,"",""],["Eigenvalue Bounds","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/02/05/eigenvalue-bounds/",0,"",""],["Local KL Divergence","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/02/02/local-kl-divergence/",0,"",""],["Quadratically Independent Monomials","jsteinhardt","2013","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2013/01/31/quadratically-independent-monomials/",0,"",""],["We are now the “Machine Intelligence Research Institute” (MIRI)","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/01/30/we-are-now-the-machine-intelligence-research-institute-miri/",0,"",""],["Yudkowsky on Logical Uncertainty","staff","2013","blog","intelligence.org","intelligence.org/2013/01/30/yudkowsky-on-logical-uncertainty/",0,"",""],["Yudkowsky on “What can we do now?”","staff","2013","blog","intelligence.org","intelligence.org/2013/01/30/yudkowsky-on-what-can-we-do-now/",0,"",""],["[LINK] NYT Article about Existential Risk from AI","anonymous","2013","blog","LessWrong","www.lesswrong.com/posts/tK37jT79YFgARZRje/link-nyt-article-about-existential-risk-from-ai",0,"",""],["CEV: a utilitarian critique","Pablo","2013","blog","LessWrong","www.lesswrong.com/posts/PnAqpopgvDGyeBCQE/cev-a-utilitarian-critique",0,"",""],["Attention-Sensitive Alerting","Eric J. Horvitz and 2 others","2013","paper","arXiv preprint","arxiv.org/abs/1301.6707",0,"",""],["AI box: AI has one shot at avoiding destruction - what might it say?","ancientcampus","2013","blog","LessWrong","www.lesswrong.com/posts/TMQY54nbmv2Pqn3ux/ai-box-ai-has-one-shot-at-avoiding-destruction-what-might-it",0,"",""],["I attempted the AI Box Experiment (and lost)","Tuxedage","2013","blog","LessWrong","www.lesswrong.com/posts/FmxhoWxvBqSxhFeJn/i-attempted-the-ai-box-experiment-and-lost",0,"",""],["2012 Winter Matching Challenge a Success!","Luke Muehlhauser","2013","blog","intelligence.org","intelligence.org/2013/01/20/2012-winter-matching-challenge-a-success/",0,"",""],["January 2013 Newsletter","staff","2013","blog","intelligence.org","intelligence.org/2013/01/09/january-2013-newsletter/",0,"",""],["New Transcript: Eliezer Yudkowsky and Massimo Pigliucci on the Intelligence Explosion","Jake","2013","blog","intelligence.org","intelligence.org/2013/01/09/new-transcript-eliezer-yudkowsky-and-massimo-pigliucci-on-the-singularity/",0,"",""],["Harsanyi's Social Aggregation Theorem and what it means for CEV","AlexMennen","2013","blog","AI Alignment Forum","www.alignmentforum.org/posts/z8afQRsH9wWsB4iMD/harsanyi-s-social-aggregation-theorem-and-what-it-means-for",0,"",""],["AI Foom Debate","Robin Hanson and Eliezer Yudkowsky","2013","report","intelligence.org","intelligence.org/ai-foom-debate/",0,"",""],["Existential Risk Prevention as Global Priority","Nick Bostrom","2013","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/abs/10.1111/1758-5899.12002",0,"",""],["General Purpose Intelligence: Arguing The Orthogonality Thesis","Stuart Armstrong","2013","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Orthogonality_Analysis_and_Metaethics-1.pdf",0,"",""],["Intelligence Explosion Microeconomics","Eliezer Yudkowsky","2013","report","intelligence.org","intelligence.org/files/IEM.pdf",0,"",""],["Knowledge and implicature: modeling language understanding as social cognition","Noah D. Goodman and Andreas Stuhlmüller","2013","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/epdf/10.1111/tops.12007",0,"",""],["Minimizing global catastrophic and existential risks from emerging technologies through international law","Grant Wilson","2013","report","gcrinstitute.org","gcrinstitute.org/minimizing-global-catastrophic-and-existential-risks-from-emerging-technologies-through-international-law/",0,"",""],["Non-pharmacological cognitive enhancement","Martin Dresler and 7 others","2013","report","ncbi.nlm.nih.gov","www.ncbi.nlm.nih.gov/pmc/articles/PMC4052735/",0,"",""],["On the Difference between Binary Prediction and True Exposure With Implications For Forecasting Tournaments and Decision Making Research","Nassim N. Taleb and Philip E. Tetlock","2013","report","stat.berkeley.edu","www.stat.berkeley.edu/~aldous/157/Papers/taleb_tetlock.pdf",0,"","forecasting"],["Power to the People: The Role of Humans in Interactive Machine Learning","Saleema Amershi and 3 others","2013","report","microsoft.com","www.microsoft.com/en-us/research/wp-content/uploads/2016/02/amershi_AIMagazine2014.pdf",0,"",""],["Risks and Mitigation Strategies for Oracle AI","Stuart Armstrong","2013","report","doi.org","doi.org/10.1007/978-3-642-31674-6_25",0,"",""],["The Ethics of Global Catastrophic Risk from Dual-Use Bioengineering","Seth D. Baum and Grant S. Wilson","2013","report","dl.begellhouse.com","www.dl.begellhouse.com/journals/6ed509641f7324e6,709fef245eef4861,06d520d747a5c0d1.html",0,"",""],["Ideal Advisor Theories and Personal CEV","lukeprog","2012","blog","LessWrong","www.lesswrong.com/posts/q9ZSXiiA7wEuRgnkS/ideal-advisor-theories-and-personal-cev",0,"",""],["Exponential Families","jsteinhardt","2012","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2012/12/21/exponential-families/",0,"",""],["December 2012 Newsletter","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/12/19/december-2012-newsletter/",0,"",""],["Algebra trick of the day","jsteinhardt","2012","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2012/12/17/algebra-trick-of-the-day/",0,"",""],["Why you must maximize expected utility","Benya","2012","blog","LessWrong","www.lesswrong.com/posts/F46jPraqp258q67nE/why-you-must-maximize-expected-utility",0,"","theory"],["Testing the AgreementMaker System in the Anatomy Task of OAEI 2012","Daniel Faria and 5 others","2012","paper","arXiv preprint","arxiv.org/abs/1212.1625",0,"","evals"],["2012 Winter Matching Challenge!","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/12/06/2012-winter-matching-challenge/",0,"",""],["Log-Linear Models","jsteinhardt","2012","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2012/12/06/log-linear-models/",0,"",""],["Once again, a reporter thinks our positions are the opposite of what they are","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/11/26/once-again-a-reporter-thinks-our-positions-are-the-opposite-of-what-they-are/",0,"",""],["An Experiment on the Connection between the DLs' Family DL<ForAllPiZero> and the Real World","Antonio Pisasale and Domenico Cantone","2012","paper","arXiv preprint","arxiv.org/abs/1211.4957",0,"",""],["A summary of the Hanson-Yudkowsky FOOM debate","Kaj_Sotala","2012","blog","LessWrong","www.lesswrong.com/posts/sPrifh6uLJQFjQJPW/a-summary-of-the-hanson-yudkowsky-foom-debate",0,"","forecasting"],["[SEQ RERUN] Brain Emulation and Hard Takeoff","MinibearRex","2012","blog","LessWrong","www.lesswrong.com/posts/xMBZTQtMSvowk3H2k/seq-rerun-brain-emulation-and-hard-takeoff",0,"","forecasting"],["How can I reduce existential risk from AI?","lukeprog","2012","blog","LessWrong","www.lesswrong.com/posts/qARBe3jBodrdPeRE6/how-can-i-reduce-existential-risk-from-ai",0,"",""],["Dating Texts without Explicit Temporal Cues","Abhimanu Kumar and 3 others","2012","paper","arXiv preprint","arxiv.org/abs/1211.2290",0,"","evals forecasting robustness"],["November 2012 Newsletter","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/11/07/november-2012-newsletter/",0,"",""],["Thinking Inside the Box: Controlling and Using an Oracle AI","Stuart Armstrong and 2 others","2012","report","link.springer.com","link.springer.com/10.1007/s11023-012-9282-2",0,"",""],["Beyond Bayesians and Frequentists","jsteinhardt","2012","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2012/10/31/beyond-bayesians-and-frequentists/",0,"",""],["Smoking lesion as a counterexample to CDT","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/eTu23XYr37prmJxa3/smoking-lesion-as-a-counterexample-to-cdt",0,"","theory"],["Cake, or death!","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/6bdb4F6Lif5AanRAd/cake-or-death",0,"",""],["Naive TDT, Bayes nets, and counterfactual mugging","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/DM87z7xQxPNMY4evK/naive-tdt-bayes-nets-and-counterfactual-mugging",0,"","theory"],["Relative Expressiveness of Defeasible Logics","Michael Maher","2012","paper","Theory and Practice of Logic Programming 12 (4-5), 793--810, 2012","arxiv.org/abs/1210.1785",0,"",""],["A Few Useful Things to Know about Machine Learning","Pedro Domingos","2012","report","homes.cs.washington.edu","homes.cs.washington.edu/~pedrod/papers/cacm12.pdf",0,"",""],["Formal Definition of AI","Dimiter Dobrev","2012","paper","International Journal \"Information Theories & Applications\",\n  vol.12, Number 3, 2005, pp.277-285","arxiv.org/abs/1209.4838",0,"",""],["September 2012 Newsletter","Jake","2012","blog","intelligence.org","intelligence.org/2012/09/21/september-2012-newsletter/",0,"",""],["RIO: Minimizing User Interaction in Ontology Debugging","Patrick Rodler and 3 others","2012","paper","arXiv preprint","arxiv.org/abs/1209.3734",0,"","robustness"],["Counterfactual Reasoning and Learning Systems","Léon Bottou and 8 others","2012","paper","arXiv preprint","arxiv.org/abs/1209.2355",0,"",""],["Counterfactual Reprogramming Decision Theory","lukeprog","2012","blog","LessWrong","www.lesswrong.com/posts/kEhSRsdhK6Dn9in7k/counterfactual-reprogramming-decision-theory",0,"","theory"],["Decision Theories, Part 3.75: Hang On, I Think This Works After All","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/X9vT3o3MmtWoRRKkm/decision-theories-part-3-75-hang-on-i-think-this-works-after",0,"","theory"],["A model of UDT with a concrete prior over logical statements","Benya","2012","blog","LessWrong","www.lesswrong.com/posts/PgKADaJE4ERjtMtP9/a-model-of-udt-with-a-concrete-prior-over-logical-statements",0,"","theory"],["Decision Theories, Part 3.5: Halt, Melt and Catch Fire","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/ShD7EHb4HmPgfveim/decision-theories-part-3-5-halt-melt-and-catch-fire",0,"","theory"],["Machine Learning A Probabilistic Perspective","Kevin Murphy","2012","report","mitpress.mit.edu","mitpress.mit.edu/9780262018029/",0,"",""],["AI timeline prediction data","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/Q6oWinLaKXmGNWGLy/ai-timeline-prediction-data",0,"","forecasting"],["August 2012 Newsletter","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/08/21/august-2012-newsletter/",0,"",""],["AI timeline predictions: are we getting better?","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/47ci9ixyEbGKWENwR/ai-timeline-predictions-are-we-getting-better",0,"","forecasting"],["Solving the two envelopes problem","rstarkov","2012","blog","LessWrong","www.lesswrong.com/posts/GezzauYzGTkcwgkA7/solving-the-two-envelopes-problem",0,"","theory"],["July 2012 Newsletter","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/08/06/july-2012-newsletter/",0,"",""],["2012 Summer Singularity Challenge Success!","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/07/30/2012-summer-singularity-challenge-success/",0,"",""],["Reasoning about Agent Programs using ATL-like Logics","Nitin Yadav and Sebastian Sardina","2012","paper","In Proceedings of the European Conference on Logics in Artificial\n  Intelligence (JELIA), volume 7519 of LNCS, pages 437-449, 2012","arxiv.org/abs/1207.3874",0,"","agents"],["Why could you be optimistic that the Singularity is Near?","gwern","2012","blog","LessWrong","www.lesswrong.com/posts/veEumGEQAsDknw9PC/why-could-you-be-optimistic-that-the-singularity-is-near",0,"","forecasting"],["Safe exploration in markov decision processes","Teodor Mihai Moldovan and Pieter Abbeel","2012","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~pabbeel/papers/MoldovanAbbeel_ICML2012full.pdf",0,"",""],["2012 Summer Singularity Challenge","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/07/03/summer-challenge/",0,"",""],["Improving neural networks by preventing co-adaptation of feature detectors","Geoffrey E. Hinton and 4 others","2012","paper","arXiv preprint","arxiv.org/abs/1207.0580",0,"","benchmarks"],["Introduction to the Theory of Computation, Chapters 1-5,7 (alternatively to Ullman and Hopcroft)","Michael Sipser","2012","report","fuuu.be","fuuu.be/polytech/INFOF408/Introduction-To-The-Theory-Of-Computation-Michael-Sipser.pdf",0,"",""],["Bounded versions of Gödel's and Löb's theorems","cousin_it","2012","blog","LessWrong","www.lesswrong.com/posts/z7SuGwxTBnQm8uFq4/bounded-versions-of-goedel-s-and-loeb-s-theorems",0,"","theory"],["Representation Learning: A Review and New Perspectives","Yoshua Bengio","2012","paper","arXiv preprint","arxiv.org/abs/1206.5538",0,"","robustness"],["Apprenticeship Learning using Inverse Reinforcement Learning and Gradient Methods","Gergely Neu and Csaba Szepesvari","2012","paper","arXiv preprint","arxiv.org/abs/1206.5264",0,"","policy"],["Imitation Learning with a Value-Based Prior","Umar Syed and Robert E. Schapire","2012","paper","arXiv preprint","arxiv.org/abs/1206.5290",0,"","policy"],["Near-Optimal BRL using Optimistic Local Transitions","Mauricio Araya and 2 others","2012","paper","arXiv preprint","arxiv.org/abs/1206.4613",0,"",""],["Machine Intelligence Research Institute Progress Report, May 2012","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/06/16/singularity-institute-progress-report-may-2012/",0,"",""],["[Link] FreakoStats and CEV","Filipe","2012","blog","LessWrong","www.lesswrong.com/posts/NM9tjAAmYQeGyPdoY/link-freakostats-and-cev",0,"",""],["List of Problems That Motivated UDT","Wei Dai","2012","blog","LessWrong","www.lesswrong.com/posts/4kvaocbkDDS2AMoPG/list-of-problems-that-motivated-udt",0,"","theory"],["Problematic Problems for TDT","drnickbone","2012","blog","LessWrong","www.lesswrong.com/posts/3GyQXTy2WhYcaBgS2/problematic-problems-for-tdt",0,"","theory"],["General purpose intelligence: arguing the Orthogonality thesis","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/nvKZchuTW8zY6wvAj/general-purpose-intelligence-arguing-the-orthogonality",0,"","instrumental-convergence"],["Machine Intelligence Research Institute Progress Report, April 2012","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/05/08/singularity-institute-progress-report-april-2012/",0,"",""],["The Superintelligent Will: Motivation and Instrumental Rationality In Advanced Intelligent Agents","Nick Bostrom","2012","report","nickbostrom.com","www.nickbostrom.com/superintelligentwill.pdf",0,"","agents"],["Non-orthogonality implies uncontrollable superintelligence","Stuart_Armstrong","2012","blog","LessWrong","www.lesswrong.com/posts/npZMkydRMqAqMqbFb/non-orthogonality-implies-uncontrollable-superintelligence",0,"",""],["Logical Uncertainty as Probability","gRR","2012","blog","LessWrong","www.lesswrong.com/posts/C5x8GiDhiaEpu54jS/logical-uncertainty-as-probability",0,"","theory"],["Stanovich on CEV","lukeprog","2012","blog","LessWrong","www.lesswrong.com/posts/GzYu2acxWL6pZyzyc/stanovich-on-cev",0,"",""],["Hofstadter's Superrationality","gwern","2012","blog","LessWrong","www.lesswrong.com/posts/pPXX56Htw5CLekAib/hofstadter-s-superrationality",0,"","theory"],["Decision Theories: A Semi-Formal Analysis, Part III","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/AMwzjjvFxEgxvL7xe/decision-theories-a-semi-formal-analysis-part-iii",0,"","theory"],["Detecting lateral genetic material transfer","C. Calderón and 3 others","2012","paper","arXiv preprint","arxiv.org/abs/1204.2601",0,"",""],["Decision Theories: A Semi-Formal Analysis, Part II","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/TxDcvtn2teAMobG2Z/decision-theories-a-semi-formal-analysis-part-ii",0,"","theory"],["Machine Intelligence Research Institute Progress Report, March 2012","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/04/06/singularity-institute-progress-report-march-2012/",0,"",""],["Making machine learning models interpretable","Alfredo Vellido and 2 others","2012","report","citeseerx.ist.psu.edu","citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.431.5382&rep=rep1&type=pdf",0,"","interpretability"],["Common mistakes people make when thinking about decision theory","cousin_it","2012","blog","LessWrong","www.lesswrong.com/posts/gkAecqbuPw4iggiub/common-mistakes-people-make-when-thinking-about-decision",0,"","theory"],["An example of self-fulfilling spurious proofs in UDT","cousin_it","2012","blog","LessWrong","www.lesswrong.com/posts/2GebvAXXfRMTjY2g7/an-example-of-self-fulfilling-spurious-proofs-in-udt",0,"","theory"],["Decision Theories: A Semi-Formal Analysis, Part I","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/2JdvZw3CXzafxQugN/decision-theories-a-semi-formal-analysis-part-i",0,"","theory"],["Modest Superintelligences","Wei Dai","2012","blog","LessWrong","www.lesswrong.com/posts/KuBMKQnAsYBGP4rkZ/modest-superintelligences",0,"",""],["A Problem About Bargaining and Logical Uncertainty","Wei Dai","2012","blog","LessWrong","www.lesswrong.com/posts/oZwxY88NCCHffJuxM/a-problem-about-bargaining-and-logical-uncertainty",0,"","theory"],["Muehlhauser-Goertzel Dialogue, Part 1","lukeprog","2012","blog","LessWrong","www.lesswrong.com/posts/TpNRpncLBAzddBnRB/muehlhauser-goertzel-dialogue-part-1",0,"","forecasting"],["Decision Theories: A Less Wrong Primer","orthonormal","2012","blog","LessWrong","www.lesswrong.com/posts/af9MjBqF2hgu3EN6r/decision-theories-a-less-wrong-primer",0,"","theory"],["Ambiguous Language and Differences in Beliefs","Joseph Y. Halpern and Willemien Kets","2012","paper","arXiv preprint","arxiv.org/abs/1203.0699",0,"","agents"],["Machine Intelligence Research Institute Progress Report, February 2012","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/03/03/singularity-institute-progress-report-february-2012/",0,"",""],["Can Intelligence Explode?","Marcus Hutter","2012","paper","Journal of Consciousness Studies, 19:1-2 (2012) 143-166","arxiv.org/abs/1202.6177",0,"",""],["Troubles With CEV Part1 - CEV Sequence","diegocaleiro","2012","blog","LessWrong","www.lesswrong.com/posts/pR5Wn7bJQWRPYNGsF/troubles-with-cev-part1-cev-sequence",0,"",""],["Troubles With CEV Part2 - CEV Sequence","diegocaleiro","2012","blog","LessWrong","www.lesswrong.com/posts/CCN5GjFnhsYiNRDCg/troubles-with-cev-part2-cev-sequence",0,"",""],["Superintelligent AGI in a box - a question.","Dmytry","2012","blog","LessWrong","www.lesswrong.com/posts/A4EBPx5htiuk22X4C/superintelligent-agi-in-a-box-a-question",0,"",""],["2011-2012 Winter Fundraiser Completed","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/02/20/2011-2012-winter-fundraiser-completed/",0,"",""],["Machine Intelligence Research Institute Progress Report, January 2012","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/02/05/singularity-institute-progress-report-january-2012/",0,"",""],["Formulas of arithmetic that behave like decision agents","Nisan","2012","blog","LessWrong","www.lesswrong.com/posts/yX9pMZik7r38da7Fc/formulas-of-arithmetic-that-behave-like-decision-agents",0,"","agents theory"],["Empowerment for Continuous Agent-Environment Systems","Tobias Jung and 2 others","2012","paper","Adaptive Behavior 19(1),2011","arxiv.org/abs/1201.6583",0,"","agents forecasting"],["Is risk aversion really irrational ?","kilobug","2012","blog","LessWrong","www.lesswrong.com/posts/ecbpjmxc833roBxj3/is-risk-aversion-really-irrational",0,"","theory"],["AI Box Log","Dorikka","2012","blog","LessWrong","www.lesswrong.com/posts/Y7uR5WqnoG629JgLn/ai-box-log",0,"",""],["Machine Intelligence Research Institute Progress Report, December 2011","Luke Muehlhauser","2012","blog","intelligence.org","intelligence.org/2012/01/16/singularity-institute-progress-report-december-2011/",0,"",""],["Q&A #2 with Luke Muehlhauser, Machine Intelligence Research Institute Executive Director","Louie Helm","2012","blog","intelligence.org","intelligence.org/2012/01/12/qa-2-with-luke-muehlhauser-singularity-institute-executive-director/",0,"",""],["Avoiding Unintended AI Behaviors","Bill Hibbard","2012","report","link.springer.com","link.springer.com/10.1007/978-3-642-35506-6_12",0,"",""],["ImageNet Classification with Deep Convolutional Neural Networks","Alex Krizevsky and 2 others","2012","report","papers.nips.cc","papers.nips.cc/paper/4824-imagenet-classification-with-deep-convolutional-neural-networks.pdf",0,"",""],["Logical Prior Probability","Abram Demski","2012","report","link.springer.com","link.springer.com/10.1007/978-3-642-35506-6_6",0,"",""],["Space-Time Embedded Intelligence","Laurent Orseau and Mark Ring","2012","report","agi-conference.org","agi-conference.org/2012/wp-content/uploads/2012/12/paper_76.pdf",0,"",""],["The Singularity and Machine Ethics","Luke Muehlhauser and Louie Helm","2012","report","link.springer.com","link.springer.com/10.1007/978-3-642-32560-1_6",0,"",""],["The Superintelligent Will: Motivation and Instrumental Rationality in Advanced Artificial Agents","Nick Bostrom","2012","report","nickbostrom.com","nickbostrom.com/superintelligentwill.pdf",0,"","agents"],["Thinking Inside the Box: Controlling and Using an Oracle AI","Stuart Armstrong and 2 others","2012","report","link.springer.com","link.springer.com/article/10.1007/s11023-012-9282-2",0,"",""],["2011 Machine Intelligence Research Institute Winter Fundraiser","Louie Helm","2011","blog","intelligence.org","intelligence.org/2011/12/27/2011-singularity-institute-winter-fundraiser/",0,"",""],["Measures, Risk, Death, and War","Vaniver","2011","blog","LessWrong","www.lesswrong.com/posts/KgWticBMH2MxgdmYc/measures-risk-death-and-war",0,"","theory"],["A model of UDT with a halting oracle","cousin_it","2011","blog","LessWrong","www.lesswrong.com/posts/Bj244uWzDBXvE2N2S/a-model-of-udt-with-a-halting-oracle",0,"","theory"],["Compressing Reality to Math","Vaniver","2011","blog","LessWrong","www.lesswrong.com/posts/2FXtpdzx6uoNRZXjS/compressing-reality-to-math",0,"","theory"],["CEV-inspired models","Stuart_Armstrong","2011","blog","LessWrong","www.lesswrong.com/posts/PBHtYurAxfm6iEpqv/cev-inspired-models",0,"",""],["5 Axioms of Decision Making","Vaniver","2011","blog","LessWrong","www.lesswrong.com/posts/zFQQEkx4c6bxdshr4/5-axioms-of-decision-making",0,"","theory"],["Where do selfish values come from?","Wei Dai","2011","blog","LessWrong","www.lesswrong.com/posts/Nz62ZurRkGPigAxMK/where-do-selfish-values-come-from",0,"",""],["Model-based utility functions","Bill Hibbard","2011","paper","Journal of Artificial General Intelligence 3(1) 1-24, 2012","arxiv.org/abs/1111.3934",0,"","agents"],["Anthropic decision theory I: Sleeping beauty and selflessness","Stuart_Armstrong","2011","blog","AI Alignment Forum","www.alignmentforum.org/posts/svhbnSdxW3XmFXXTK/anthropic-decision-theory-i-sleeping-beauty-and-selflessness",0,"","theory"],["Anthropic decision theory","Stuart Armstrong","2011","paper","arXiv preprint","arxiv.org/abs/1110.6437",0,"","agents theory"],["In favour of a selective CEV initial dynamic","anonymous","2011","blog","LessWrong","www.lesswrong.com/posts/u8isNgN7rRYBZ35rQ/in-favour-of-a-selective-cev-initial-dynamic",0,"",""],["Multi-Issue Negotiation with Deadlines","S. S. Fatima and 2 others","2011","paper","Journal Of Artificial Intelligence Research, Volume 27, pages\n  381-417, 2006","arxiv.org/abs/1110.2765",0,"","agents"],["Formal verification of hybrid systems","Rajeev Alur","2011","report","dl.acm.org","dl.acm.org/citation.cfm?doid=2038642.2038685",0,"",""],["Global Catastrophic Risks","Nick Bostrom and Milan M. Cirkovic","2011","report","goodreads.com","www.goodreads.com/book/show/2659696-global-catastrophic-risks",0,"",""],["Interview with New MIRI Research Fellow Luke Muehlhauser","Louie Helm","2011","blog","intelligence.org","intelligence.org/2011/09/15/interview-with-new-singularity-institute-research-fellow-luke-muehlhuaser-september-2011/",0,"",""],["Decision Theory Paradox: Answer Key","orthonormal","2011","blog","LessWrong","www.lesswrong.com/posts/HxYneuv9XRit4dMRm/decision-theory-paradox-answer-key",0,"","theory"],["Consequentialism Need Not Be Nearsighted","orthonormal","2011","blog","LessWrong","www.lesswrong.com/posts/prb8raC4XGJiRWs5n/consequentialism-need-not-be-nearsighted",0,"","theory"],["2011 Summer Matching Challenge Success!","Luke Muehlhauser","2011","blog","intelligence.org","intelligence.org/2011/09/01/2011-summer-matching-challenge-success/",0,"",""],["Introduction: Open Questions in Roboethics","John P. Sullins","2011","report","link.springer.com","link.springer.com/10.1007/s13347-011-0043-6",0,"",""],["Decision Theory Paradox: PD with Three Implies Chaos?","orthonormal","2011","blog","LessWrong","www.lesswrong.com/posts/HT8jwNJ6vH7p9gaTT/decision-theory-paradox-pd-with-three-implies-chaos",0,"","theory"],["Machine Intelligence Research Institute Strategic Plan 2011","Louie Helm","2011","blog","intelligence.org","intelligence.org/2011/08/26/singularity-institute-strategic-plan-2011/",0,"",""],["New Intelligence Explosion Website","Luke Muehlhauser","2011","blog","intelligence.org","intelligence.org/2011/08/07/new-intelligence-explosion-website/",0,"",""],["Asymptotically Optimal Agents","Tor Lattimore and Marcus Hutter","2011","paper","Proc. 22nd International Conf. on Algorithmic Learning Theory\n  (ALT-2011) pages 368-382","arxiv.org/abs/1107.5537",0,"","agents"],["What's wrong with simplicity of value?","Wei Dai","2011","blog","LessWrong","www.lesswrong.com/posts/4xWz3wW2JNfup6By6/what-s-wrong-with-simplicity-of-value",0,"",""],["Announcing the $125,000 Summer Singularity Challenge","Luke Muehlhauser","2011","blog","intelligence.org","intelligence.org/2011/07/22/announcing-the-125000-summer-singularity-challenge/",0,"",""],["Some Thoughts on Singularity Strategies","Wei Dai","2011","blog","LessWrong","www.lesswrong.com/posts/73SotZnDbsYpxfnuQ/some-thoughts-on-singularity-strategies",0,"","scalable-oversight"],["AI-Box Experiment - The Acausal Trade Argument","XiXiDu","2011","blog","LessWrong","www.lesswrong.com/posts/DYcXRiJWiAtbXxNA5/ai-box-experiment-the-acausal-trade-argument",0,"",""],["Topics to discuss CEV","diegocaleiro","2011","blog","LessWrong","www.lesswrong.com/posts/TXqYCxKupcsLcXkoz/topics-to-discuss-cev",0,"",""],["Verifying Stability of Stochastic Systems","jsteinhardt","2011","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2011/07/03/verifying-stability-of-stochastic-systems/",0,"",""],["An Architectural Approach to Ensuring Consistency in Hierarchical Execution","J. E. Laird and R. E. Wray","2011","paper","Journal Of Artificial Intelligence Research, Volume 19, pages\n  355-398, 2003","arxiv.org/abs/1106.4871",0,"","agents"],["I Don't Want to Think About it Now:Decision Theory With Costly Computation","Joseph Y. Halpern and Rafael Pass","2011","paper","arXiv preprint","arxiv.org/abs/1106.2657",0,"","theory"],["An Application of Reinforcement Learning to Dialogue Strategy Selection in a Spoken Dialogue System for Email","M. A. Walker","2011","paper","Journal Of Artificial Intelligence Research, Volume 12, pages\n  387-416, 2000","arxiv.org/abs/1106.0241",0,"","evals agents"],["The Joys of Conjugate Priors","TCB","2011","blog","LessWrong","www.lesswrong.com/posts/u2gWM2poRPkBPFeLc/the-joys-of-conjugate-priors",0,"","theory"],["Example decision theory problem: \"Agent simulates predictor\"","cousin_it","2011","blog","LessWrong","www.lesswrong.com/posts/q9DbfYfFzkotno9hG/example-decision-theory-problem-agent-simulates-predictor",0,"","agents theory"],["Ontological Crises in Artificial Agents' Value Systems","Peter de Blanc","2011","paper","arXiv preprint","arxiv.org/abs/1105.3821",0,"","agents"],["Beginning resources for CEV research","lukeprog","2011","blog","LessWrong","www.lesswrong.com/posts/jN2gbDRJHTtXYSdhY/beginning-resources-for-cev-research",0,"",""],["Real-world Newcomb-like Problems","SilasBarta","2011","blog","LessWrong","www.lesswrong.com/posts/TejMdvF9XTNP5pGDR/real-world-newcomb-like-problems",0,"","theory"],["How I Lost 100 Pounds Using TDT","Zvi","2011","blog","LessWrong","www.lesswrong.com/posts/scwoBEju75C45W5n3/how-i-lost-100-pounds-using-tdt",0,"","theory"],["How to Grow a Mind: Statistics, Structure, and Abstraction","Joshua B. Tenenbaum and 3 others","2011","report","cocosci.princeton.edu","cocosci.princeton.edu/tom/papers/LabPublications/GrowMind.pdf",0,"",""],["The Urgent Meta-Ethics of Friendly Artificial Intelligence","lukeprog","2011","blog","LessWrong","www.lesswrong.com/posts/TKdpSzmcezNbfmGAy/the-urgent-meta-ethics-of-friendly-artificial-intelligence",0,"",""],["Useful Math","jsteinhardt","2011","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2011/01/23/useful-math/",0,"",""],["Tallinn-Evans Challenge Grant Success!","Louie Helm","2011","blog","intelligence.org","intelligence.org/2011/01/20/tallinn-evans-challenge-grant-success/",0,"",""],["[LINK] What should a reasonable person believe about the Singularity?","Kaj_Sotala","2011","blog","LessWrong","www.lesswrong.com/posts/2XZju58cP82Fv776N/link-what-should-a-reasonable-person-believe-about-the",0,"","forecasting"],["A Prima Facie Duty Approach to Machine Ethics","Susan Leigh Anderson and Michael Anderson","2011","report","cambridge.org","www.cambridge.org/core/product/identifier/CBO9780511978036A041/type/book_part",0,"",""],["Bayesian Theory of Mind: Modeling Joint Belief-Desire Attribution","Chris L. Baker and 2 others","2011","report","aiweb.cs.washington.edu","aiweb.cs.washington.edu/research/projects/aiweb/media/papers/cogsci2011.pdf",0,"",""],["Complex Value Systems in Friendly AI","Eliezer Yudkowsky","2011","report","link.springer.com","link.springer.com/10.1007/978-3-642-22887-2_48",0,"",""],["Delusion, Survival, and Intelligent Agents","Mark Ring and Laurent Orseau","2011","report","link.springer.com","link.springer.com/10.1007/978-3-642-22887-2_2",0,"","agents"],["Delusion, Survival, and Intelligent Agents","Mark Ring and Laurent Orseau","2011","report","link.springer.com","link.springer.com/chapter/10.1007%2F978-3-642-22887-2_2",0,"","agents"],["How long until human-level AI? Results from an expert assessment","Seth D. Baum and 2 others","2011","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0040162510002106",0,"",""],["Knows What It Knows: A Framework For Self-Aware Learning","Lihong Li and 2 others","2011","report","machinelearning.org","www.machinelearning.org/archive/icml2008/papers/627.pdf",0,"","situational-awareness"],["Learning What to Value","Daniel Dewey","2011","report","danieldewey.net","www.danieldewey.net/learning-what-to-value.pdf",0,"",""],["Learning What to Value","Daniel Dewey","2011","report","link.springer.com","link.springer.com/10.1007/978-3-642-22887-2_35",0,"",""],["Online Learning Survey","Shai Shalev-Schwartz","2011","report","cs.huji.ac.il","www.cs.huji.ac.il/~shais/papers/OLsurvey.pdf",0,"",""],["Risks of Astronomical Future Suﬀering","Brian Tomasik","2011","report","longtermrisk.org","longtermrisk.org/files/risks-of-astronomical-future-suffering.pdf",0,"",""],["Yudkowsky-Hanson Jane Street Debate 2011","Eliezer Yudkowsky and Robin Hanson","2010","report","drive.google.com","drive.google.com/file/d/1vgZqoUWMOL4WRjt2lILflJWVsk2kEj0H/view?usp=share_link",0,"",""],["Looking for plausibility","Wan Ahmad Tajuddin Wan Abdullah","2010","paper","arXiv preprint","arxiv.org/abs/1012.5705",0,"",""],["Ontology-based Queries over Cancer Data","Alejandra Gonzalez-Beltran and 2 others","2010","paper","arXiv preprint","arxiv.org/abs/1012.5506",0,"","evals deception"],["Two questions about CEV that worry me","cousin_it","2010","blog","LessWrong","www.lesswrong.com/posts/wLmxiXfpLjiTBiT2j/two-questions-about-cev-that-worry-me",0,"",""],["Announcing the Tallinn-Evans $125,000 Singularity Challenge","Louie Helm","2010","blog","intelligence.org","intelligence.org/2010/12/21/announcing-the-tallinn-evans-125000-singularity-holiday-challenge/",0,"",""],["Solve Psy-Kosh's non-anthropic problem","cousin_it","2010","blog","LessWrong","www.lesswrong.com/posts/YZzoWGCJsoRBBbmQg/solve-psy-kosh-s-non-anthropic-problem",0,"","theory"],["Cryptographic Boxes for Unfriendly AI","paulfchristiano","2010","blog","AI Alignment Forum","www.alignmentforum.org/posts/2Wf3R4NZ77CLczLL2/cryptographic-boxes-for-unfriendly-ai",0,"",""],["Does TDT pay in Counterfactual Mugging?","Bongo","2010","blog","LessWrong","www.lesswrong.com/posts/6eGw3CJDDqwrSYiHu/does-tdt-pay-in-counterfactual-mugging",0,"","theory"],["Criticisms of CEV (request for links)","Kevin","2010","blog","LessWrong","www.lesswrong.com/posts/ApECAMi7fuexNaQ4K/criticisms-of-cev-request-for-links",0,"",""],["Another attempt to explain UDT","cousin_it","2010","blog","LessWrong","www.lesswrong.com/posts/zztyZ4SKy7suZBpbk/another-attempt-to-explain-udt",0,"","theory"],["References & Resources for LessWrong","XiXiDu","2010","blog","LessWrong","www.lesswrong.com/posts/TNHQLZK5pHbxdnz4e/references-and-resources-for-lesswrong",0,"","theory"],["Generalizing Across Categories","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/10/02/generalizing-across-categories/",0,"",""],["Uncertain Observations","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/09/18/uncertain-observations/",0,"",""],["Nobody Understands Probability","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/09/13/nobody-understands-probability/",0,"",""],["Humans are not automatically strategic","AnnaSalamon","2010","blog","LessWrong","www.lesswrong.com/posts/PBRWb2Em5SNeWYwwB/humans-are-not-automatically-strategic",0,"","goodharts-law"],["Controlling Constant Programs","Vladimir_Nesov","2010","blog","LessWrong","www.lesswrong.com/posts/gZbHSWcLvj7ZopSas/controlling-constant-programs",0,"","theory"],["Anthropomorphic AI and Sandboxed Virtual Universes","jacob_cannell","2010","blog","LessWrong","www.lesswrong.com/posts/5P6sNqP7N9kSA97ao/anthropomorphic-ai-and-sandboxed-virtual-universes",0,"",""],["Least Squares and Fourier Analysis","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/08/22/least-squares-and-fourier-analysis/",0,"",""],["Newcomb's Problem: A problem for Causal Decision Theories","anonymous","2010","blog","LessWrong","www.lesswrong.com/posts/WTA6vmYdQCzTFT4WZ/newcomb-s-problem-a-problem-for-causal-decision-theories",0,"","theory"],["An introduction to decision theory","anonymous","2010","blog","LessWrong","www.lesswrong.com/posts/cRrM8LPf9waAd4uiL/an-introduction-to-decision-theory",0,"","theory"],["What a reduction of \"could\" could look like","cousin_it","2010","blog","LessWrong","www.lesswrong.com/posts/dC3rxrMkYKLfgTYEa/what-a-reduction-of-could-could-look-like",0,"","theory"],["AI cooperation in practice","cousin_it","2010","blog","LessWrong","www.lesswrong.com/posts/TNfx89dh5KkcKrvho/ai-cooperation-in-practice",0,"","theory"],["Linear Control Theory: Part I","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/07/17/linear-control-theory-part-i/",0,"",""],["The Underwater Cartpole","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/06/26/the-underwater-cartpole/",0,"",""],["Linear Control Theory: Part 0","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/06/20/linear-control/",0,"",""],["What if AI doesn't quite go FOOM?","Mass_Driver","2010","blog","LessWrong","www.lesswrong.com/posts/wv6a9kA6EApiYD5sL/what-if-ai-doesn-t-quite-go-foom",0,"","forecasting"],["Robotics","jsteinhardt","2010","blog","jsteinhardt.wordpress.com","jsteinhardt.wordpress.com/2010/06/18/robotics/",0,"",""],["ToLeRating UR-STD","Jan Feyereisl and Uwe Aickelin","2010","paper","Proceedings of the 2nd International Conference on Emerging\n  Security Information, Systems and Technologies, Cap Esterel, France, p\n  287-293, 2008","arxiv.org/abs/1006.1563",0,"","monitoring"],["Hacking the CEV for Fun and Profit","Wei Dai","2010","blog","LessWrong","www.lesswrong.com/posts/HW5Q9cW9sgk4yCffd/hacking-the-cev-for-fun-and-profit",0,"",""],["How to Explain Individual Classification Decisions","David Baehrens and 5 others","2010","report","jmlr.org","www.jmlr.org/papers/volume11/baehrens10a/baehrens10a.pdf",0,"",""],["What is Wei Dai's Updateless Decision Theory?","AlephNeil","2010","blog","LessWrong","www.lesswrong.com/posts/5WEoM3RCxN2cQEdzY/what-is-wei-dai-s-updateless-decision-theory",0,"","theory"],["Is Google Paperclipping the Web? The Perils of Optimization by Proxy in Social Systems","Alexandros","2010","blog","LessWrong","www.lesswrong.com/posts/fTu69HzLSXqWgj9ib/is-google-paperclipping-the-web-the-perils-of-optimization",0,"","goodharts-law"],["Only humans can have human values","PhilGoetz","2010","blog","LessWrong","www.lesswrong.com/posts/cAPCCJjggjZPxxcKh/only-humans-can-have-human-values",0,"",""],["Rational Value of Information Estimation for Measurement Selection","David Tolpin and Solomon Eyal Shimony","2010","paper","arXiv preprint","arxiv.org/abs/1003.5305",0,"","evals"],["Newcomb's problem happened to me","Academian","2010","blog","LessWrong","www.lesswrong.com/posts/6fvzjL4duMsWXswKf/newcomb-s-problem-happened-to-me",0,"","theory"],["The Importance of Goodhart's Law","blogospheroid","2010","blog","LessWrong","www.lesswrong.com/posts/YtvZxRpZjcFNwJecS/the-importance-of-goodhart-s-law",0,"","goodharts-law"],["The Blackmail Equation","Stuart_Armstrong","2010","blog","LessWrong","www.lesswrong.com/posts/Ljy3CSwTFPEpnGLLJ/the-blackmail-equation",0,"","theory"],["What does Newcomb's paradox teach us?","David H. Wolpert and Gregory Benford","2010","paper","arXiv preprint","arxiv.org/abs/1003.1343",0,"","deception"],["2010 Singularity Research Challenge Fulfilled!","Louie Helm","2010","blog","intelligence.org","intelligence.org/2010/03/01/2010-singularity-research-challenge-fulfilled/",0,"",""],["Agent Based Approaches to Engineering Autonomous Space Software","Louise A. Dennis and 4 others","2010","paper","EPTCS 20, 2010, pp. 63-67","arxiv.org/abs/1003.0617",0,"","agents"],["Probing the improbable: methodological challenges for risks with low probabilities and high stakes","Toby Ord and 2 others","2010","report","doi.org","doi.org/10.1080/13669870903126267",0,"","interpretability"],["Babies and Bunnies: A Caution About Evo-Psych","Alicorn","2010","blog","LessWrong","www.lesswrong.com/posts/u9pfbkZeG8mFTPNi2/babies-and-bunnies-a-caution-about-evo-psych",0,"",""],["Explicit Optimization of Global Strategy (Fixing a Bug in UDT1)","Wei Dai","2010","blog","LessWrong","www.lesswrong.com/posts/g8xh9R7RaNitKtkaa/explicit-optimization-of-global-strategy-fixing-a-bug-in",0,"","theory"],["A problem with Timeless Decision Theory (TDT)","Gary_Drescher","2010","blog","LessWrong","www.lesswrong.com/posts/NSX8RuD9tQ4uWzkk3/a-problem-with-timeless-decision-theory-tdt",0,"","theory"],["Applying utility functions to humans considered harmful","Kaj_Sotala","2010","blog","LessWrong","www.lesswrong.com/posts/vuN57BvWyT7WZ3b6p/applying-utility-functions-to-humans-considered-harmful",0,"","deception"],["The AI in a box boxes you","Stuart_Armstrong","2010","blog","LessWrong","www.lesswrong.com/posts/c5GHf2kMGhA4Tsj4g/the-ai-in-a-box-boxes-you",0,"",""],["Value Uncertainty and the Singleton Scenario","Wei Dai","2010","blog","LessWrong","www.lesswrong.com/posts/Qh6bnkxbMFz5SNeFd/value-uncertainty-and-the-singleton-scenario",0,"",""],["Risks and Mitigation Strategies for Oracle AI","Stuart Armstrong","2010","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/wp-content/uploads/Risks-and-Mitigation-Strategies-for-Oracle-AI.pdf",0,"",""],["The Quest for Artificial Intelligence","Nils J. Nilsson","2010","report","ai.stanford.edu","ai.stanford.edu/~nilsson/QAI/qai.pdf",0,"",""],["Towards State Summarization for Autonomous Robots","Daniel Brooks and 4 others","2010","report","robotics.cs.uml.edu","robotics.cs.uml.edu/fileadmin/content/publications/2010/towards_state_summarization_11-10.pdf",0,"",""],["Utility Indifference","Stuart Armstrong","2010","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/reports/2010-1.pdf",0,"",""],["A Rational Decision Maker with Ordinal Utility under Uncertainty: Optimism and Pessimism","Ji Han","2009","paper","arXiv preprint","arxiv.org/abs/0912.5073",0,"",""],["Announcing the 2010 Singularity Research Challenge","Tom McCabe","2009","blog","intelligence.org","intelligence.org/2009/12/23/announcing-the-2010-singularity-research-challenge/",0,"",""],["Action Understanding as Inverse Planning","Chris L. Baker and 2 others","2009","report","sciencedirect.com","www.sciencedirect.com/science/article/abs/pii/S0010027709001607",0,"",""],["Help or Hinder: Bayesian Models of Social Goal Inference","Tomer D. Ullman and 5 others","2009","report","dspace.mit.edu","dspace.mit.edu/handle/1721.1/61347",0,"",""],["The Sword of Good","Eliezer S. Yudkowsky","2009","blog","yudkowsky.net","www.yudkowsky.net/other/fiction/the-sword-of-good",0,"","robustness"],["What Program Are You?","RobinHanson","2009","blog","LessWrong","www.lesswrong.com/posts/5r7jgoZN6eDN7M5hK/what-program-are-you",0,"","theory"],["Quantum Russian Roulette","Christian_Szegedy","2009","blog","LessWrong","www.lesswrong.com/posts/XH9ZN8bLidtcqMxY2/quantum-russian-roulette",0,"","theory"],["The Absent-Minded Driver","Wei Dai","2009","blog","LessWrong","www.lesswrong.com/posts/GfHdNfqxe3cSCfpHL/the-absent-minded-driver",0,"","theory"],["Decision theory: Why Pearl helps reduce “could” and “would”, but still leaves us with at least three alternatives","AnnaSalamon","2009","blog","LessWrong","www.lesswrong.com/posts/miwf7qQTh2HXNnSuq/decision-theory-why-pearl-helps-reduce-could-and-would-but",0,"","theory"],["Assessing the Impact of Informedness on a Consultant's Profit","Eugen Staab and Martin Caminada","2009","paper","arXiv preprint","arxiv.org/abs/0909.0901",0,"",""],["Decision theory: Why we need to reduce “could”, “would”, “should”","AnnaSalamon","2009","blog","LessWrong","www.lesswrong.com/posts/gxxpK3eiSQ3XG3DW7/decision-theory-why-we-need-to-reduce-could-would-should",0,"","theory"],["Interactively shaping agents via human reinforcement: The TAMER framework","W. Bradley Knox and Peter Stone","2009","report","cs.utexas.edu","www.cs.utexas.edu/~pstone/Papers/bib2html-links/KCAP09-knox.pdf",0,"","agents"],["Confusion about Newcomb is confusion about counterfactuals","AnnaSalamon","2009","blog","LessWrong","www.lesswrong.com/posts/B7bMmhvaufdtxBtLW/confusion-about-newcomb-is-confusion-about-counterfactuals",0,"","theory"],["Decision theory: An outline of some upcoming posts","AnnaSalamon","2009","blog","LessWrong","www.lesswrong.com/posts/sLxFqs8fdjsPdkpLC/decision-theory-an-outline-of-some-upcoming-posts",0,"","theory"],["Timeless Decision Theory and Meta-Circular Decision Theory","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/fQv85Rd3pw789MHaX/timeless-decision-theory-and-meta-circular-decision-theory",0,"","theory"],["Ingredients of Timeless Decision Theory","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/szfxvS8nsxTgJLBHs/ingredients-of-timeless-decision-theory",0,"","theory"],["Towards a New Decision Theory","Wei Dai","2009","blog","LessWrong","www.lesswrong.com/posts/de3xjFaACCAk6imzv/towards-a-new-decision-theory",0,"","theory"],["Probabilistic Graphical Models: Principles and Techniques","Daphne Koller and Nir Friedman","2009","report","mitpress.mit.edu","mitpress.mit.edu/books/probabilistic-graphical-models",0,"",""],["Bayesian Utility: Representing Preference by Probability Measures","Vladimir_Nesov","2009","blog","AI Alignment Forum","www.alignmentforum.org/posts/kYgWmKJnqq8QkbjFj/bayesian-utility-representing-preference-by-probability",0,"",""],["Timeless Decision Theory: Problems I Can't Solve","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/c3wWnvgzdbRhNnNbQ/timeless-decision-theory-problems-i-can-t-solve",0,"","theory"],["The Strangest Thing An AI Could Tell You","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/t2NN6JwMFaqANuLqH/the-strangest-thing-an-ai-could-tell-you",0,"","deception"],["The Fixed Sum Fallacy","cousin_it","2009","blog","LessWrong","www.lesswrong.com/posts/7T5J3WM5zqcnTPofS/the-fixed-sum-fallacy",0,"","theory"],["Reasoning About Knowledge of Unawareness Revisited","Joseph Y. Halpern and Leandro Rego","2009","paper","arXiv preprint","arxiv.org/abs/0906.4321",0,"","agents"],["Shane Legg on prospect theory and computational finance","Roko","2009","blog","LessWrong","www.lesswrong.com/posts/PnhpMqMP75Dxpvar5/shane-legg-on-prospect-theory-and-computational-finance",0,"","theory"],["Bayesians vs. Barbarians","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/KsHmn6iJAEr9bACQW/bayesians-vs-barbarians",0,"","theory"],["Finding Causes of Program Output with the Java Whyline","Andrew Ko and Brad A. Myers","2009","report","faculty.washington.edu","faculty.washington.edu/ajko/papers/Ko2009JavaWhylineUI.pdf",0,"",""],["Newcomb's Problem standard positions","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/WhzCbrxG4KzFz7W4d/newcomb-s-problem-standard-positions",0,"","theory"],["Formalizing Newcomb's","cousin_it","2009","blog","LessWrong","www.lesswrong.com/posts/GZ8t3uJRPSQb2sAH3/formalizing-newcomb-s",0,"","theory"],["Rationality: Common Interest of Many Causes","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/4PPE6D635iBcGPGRy/rationality-common-interest-of-many-causes",0,"","instrumental-convergence"],["Building the information kernel and the problem of recognition","Elena S. Vishnevskaya","2009","paper","arXiv preprint","arxiv.org/abs/0903.4513",0,"",""],["Counterfactual Mugging","Vladimir_Nesov","2009","blog","LessWrong","www.lesswrong.com/posts/mg6jDEuQEjBGtibX7/counterfactual-mugging",0,"","theory"],["You May Already Be A Sinner","Scott Alexander","2009","blog","LessWrong","www.lesswrong.com/posts/Cq45AuedYnzekp3LX/you-may-already-be-a-sinner",0,"","theory"],["Markets are Anti-Inductive","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/h24JGbmweNpWZfBkM/markets-are-anti-inductive",0,"","goodharts-law"],["Introducing Myself","Michael Vassar","2009","blog","intelligence.org","intelligence.org/2009/02/16/introducing-myself/",0,"",""],["Value is Fragile","Eliezer Yudkowsky","2009","blog","LessWrong","www.lesswrong.com/posts/GNnHHmm8EzePmKzPk/value-is-fragile",0,"",""],["A survey on transfer learning","Sinno Jialin Pan and Qiang Yang","2009","report","cse.ust.hk","www.cse.ust.hk/~qyang/Docs/2009/tkde_transfer_learning.pdf",0,"",""],["Pascal’s Mugging","Nick Bostrom","2009","report","nickbostrom.com","www.nickbostrom.com/papers/pascal.pdf",0,"",""],["Why I Want to be a Posthuman when I Grow Up","Nick Bostrom","2009","report","doi.org","doi.org/10.1007/978-1-4020-8852-0_8",0,"",""],["What I Think, If Not Why","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/z3kYdw54htktqt9Jb/what-i-think-if-not-why",0,"","forecasting"],["True Sources of Disagreement","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/3pyLbH3BqevetQros/true-sources-of-disagreement",0,"","forecasting"],["Hard Takeoff","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/tjH8XPxAnr6JRbh7k/hard-takeoff",0,"","forecasting"],["Singletons Rule OK","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/rSTpxugJxFPoRMkGW/singletons-rule-ok",0,"","governance"],["Engelbart: Insufficiently Recursive","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/NCb28Xdv7xDajtqtS/engelbart-insufficiently-recursive",0,"","forecasting"],["...Recursion, Magic","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/rJLviHqJMTy8WQkow/recursion-magic",0,"","forecasting"],["Cascades, Cycles, Insight...","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/dq3KsCsqNotWc8nAK/cascades-cycles-insight",0,"","forecasting"],["Surprised by Brains","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/XQirei3crsLxsCQoi/surprised-by-brains",0,"","forecasting"],["Visualizing Data Using t-SNE","Laurens van der Maaten and Geoff Hinton","2008","report","jmlr.org","www.jmlr.org/papers/volume9/vandermaaten08a/vandermaaten08a.pdf",0,"",""],["Ends Don't Justify Means (Among Humans)","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/K9ZaZXDnL3SEmYZqB/ends-don-t-justify-means-among-humans",0,"","deception theory"],["AIs and Gatekeepers Unite!","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/CoEtbtMTcPczTiPuX/ais-and-gatekeepers-unite",0,"",""],["Probing the Improbable: Methodological Challenges for Risks with Low Probabilities and High Stakes","Toby Ord and 2 others","2008","report","researchgate.net","www.researchgate.net/publication/251508313_Probing_the_Improbable_Methodological_Challenges_for_Risks_with_Low_Probabilities_and_High_Stakes",0,"","interpretability"],["Dreams of Friendliness","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/wKnwcjJGriTS9QxxL/dreams-of-friendliness",0,"",""],["The Gift We Give To Tomorrow","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/pGvyqAQw6yqTjpKf4/the-gift-we-give-to-tomorrow",0,"",""],["Artificial Intelligence as a positive and negative factor in global risk","Eliezer Yudkowsky","2008","report","oxford.universitypressscholarship.com","oxford.universitypressscholarship.com/view/10.1093/oso/9780198570509.001.0001/isbn-9780198570509-book-part-21",0,"",""],["Cognitive biases potentially affecting judgement of global risks","Eliezer Yudkowsky","2008","report","oxford.universitypressscholarship.com","oxford.universitypressscholarship.com/view/10.1093/oso/9780198570509.001.0001/isbn-9780198570509-book-part-9",0,"",""],["Economics of the singularity","Robin Hanson","2008","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/4531461/",0,"",""],["Explaining classifications for individual instances","M. Robnik-Sikonja and I. Kononenko","2008","report","researchgate.net","www.researchgate.net/publication/3297901_Explaining_Classifications_For_Individual_Instances",0,"",""],["That Alien Message","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/5wMcKNAwB6X4mp9og/that-alien-message",0,"",""],["A Comprehensive Survey of Multiagent Reinforcement Learning","Lucian Busoniu and 2 others","2008","report","researchgate.net","www.researchgate.net/publication/3421909_A_Comprehensive_Survey_of_Multiagent_Reinforcement_Learning",0,"","agents"],["Idiotypic Immune Networks in Mobile Robot Control","Amanda Whitbrook and 2 others","2008","paper","IEEE Transactions on Systems, Man and Cybernetics, Part B, 37(6),\n  1581- 1598, 2007","arxiv.org/abs/0803.2981",0,"","deception"],["The Basic AI Drives","Stephen Omohundro","2008","report","selfawaresystems.files.wordpress.com","selfawaresystems.files.wordpress.com/2008/01/ai_drives_final.pdf",0,"","robustness"],["Newcomb's Problem and Regret of Rationality","Eliezer Yudkowsky","2008","blog","LessWrong","www.lesswrong.com/posts/6ddcsdA2c2XpNpE5x/newcomb-s-problem-and-regret-of-rationality",0,"","theory"],["Global Catastrophic Risks Survey","Anders Sandberg and Nick Bostrom","2008","report","fhi.ox.ac.uk","www.fhi.ox.ac.uk/reports/2008-1.pdf",0,"",""],["Principles of Model Checking, Chapter 1-7,10","Christel Baier and Joost-Pieter Katoen","2008","report","researchgate.net","www.researchgate.net/publication/220690719_Principles_of_Model_Checking",0,"",""],["Pascal's Mugging: Tiny Probabilities of Vast Utilities","Eliezer Yudkowsky","2007","blog","LessWrong","www.lesswrong.com/posts/a5JAiTdytou3Jg749/pascal-s-mugging-tiny-probabilities-of-vast-utilities",0,"","theory"],["Three Major Singularity Schools","Eliezer Yudkowsky","2007","blog","intelligence.org","intelligence.org/2007/09/30/three-major-singularity-schools/",0,"",""],["The Power of Intelligence","Eliezer Yudkowsky","2007","blog","intelligence.org","intelligence.org/2007/07/10/the-power-of-intelligence/",0,"",""],["Sleeping Beauty and Self-location: A Hybrid Model","Nick Bostrom","2007","report","link.springer.com","link.springer.com/10.1007/s11229-006-9010-7",0,"",""],["If multi-agent learning is the answer, what is the question?","Yoav Shoham and 2 others","2007","report","linkinghub.elsevier.com","linkinghub.elsevier.com/retrieve/pii/S0004370207000495",0,"","agents"],["The Myth of the Rational Voter: Why Democracies Choose Bad Policies","Bryan Caplan","2007","report","goodreads.com","www.goodreads.com/book/show/698866.The_Myth_of_the_Rational_Voter",0,"",""],["Universal Algorithmic Intelligence: A mathematical top->down approach","Marcus Hutter","2007","paper","In Artificial General Intelligence, Springer (2007) 227-290","arxiv.org/abs/cs/0701125",0,"","agents theory"],["Bayesian Inverse Reinforcement Learning","Deepak Ramachandran","2007","report","ijcai.org","www.ijcai.org/Proceedings/07/Papers/416.pdf",0,"",""],["Artificial General Intelligence","Ben Goertzel and Cassio Pennachin","2007","report","link.springer.com","link.springer.com/10.1007/978-3-540-68677-4",0,"",""],["Modelling Morality with Prospective Logic","Luís Moniz Pereira and Ari Saptawijaya","2007","report","link.springer.com","link.springer.com/10.1007/978-3-540-77002-2_9",0,"",""],["Computational Models of Ethical Reasoning: Challenges, Initial Steps, and Future Directions","B.M. McLaren","2006","report","ieeexplore.ieee.org","ieeexplore.ieee.org/document/1667950/",0,"",""],["A Decision-Making Support System Based on Know-How","V. V. Kryssanov and 3 others","2006","paper","CIRP Journal of Manufacturing Systems. 1998, Vol. 27, No.4,\n  427-432","arxiv.org/abs/cs/0606010",0,"",""],["Twelve Virtues of Rationality","Eliezer S. Yudkowsky","2006","blog","yudkowsky.net","www.yudkowsky.net/rational/virtues",0,"",""],["A Formal Measure of Machine Intelligence","Shane Legg and Marcus Hutter","2006","paper","Proc. 15th Annual Machine Learning Conference of {B}elgium and The\n  Netherlands (Benelearn 2006) pages 73-80","arxiv.org/abs/cs/0605024",0,"",""],["Assuring the Behavior of Adaptive Agents","Diana F. Spears","2006","report","link.springer.com","link.springer.com/chapter/10.1007/1-84628-271-3_8",0,"","agents"],["Building Explainable Artificial Intelligence Systems","Mark G. Core and 5 others","2006","report","aaai.org","www.aaai.org/Papers/AAAI/2006/AAAI06-293.pdf",0,"",""],["Can Machine Learning Be Secure?","Marco Barreno and 4 others","2006","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~tygar/papers/Machine_Learning_Security/asiaccs06.pdf",0,"",""],["Maximum Margin Planning","Nathan D. Ratliff and 2 others","2006","report","martin.zinkevich.org","martin.zinkevich.org/publications/maximummarginplanning.pdf",0,"",""],["Transparency and Socially Guided Machine Learning","Andrea L. Thomaz and Cynthia Breazeal","2006","report","cc.gatech.edu","www.cc.gatech.edu/~athomaz/papers/ThomazBreazeal-ICDL06.pdf",0,"","interpretability"],["How unlikely is a doomsday catastrophe?","Max Tegmark and Nick Bostrom","2005","paper","arXiv preprint","arxiv.org/abs/astro-ph/0512204",0,"",""],["Evidence with Uncertain Likelihoods","Joseph Y. Halpern and Riccardo Pucella","2005","paper","arXiv preprint","arxiv.org/abs/cs/0510079",0,"","agents robustness"],["The Fable of the Dragon-Tyrant","Nick Bostrom","2005","report","nickbostrom.com","nickbostrom.com/fable/dragon",0,"",""],["Explainable Artificial Intelligence for Training and Tutoring","H. Chad Lane and 4 others","2005","report","researchgate.net","www.researchgate.net/publication/221297503_Explainable_Artificial_Intelligence_for_Training_and_Tutoring",0,"",""],["Power laws, Pareto distributions and Zipf’s law","M. E. J. Newman","2004","paper","Contemporary Physics 46, 323-351 (2005)","arxiv.org/abs/cond-mat/0412004",0,"",""],["Intelligent Machinery, A Heretical Theory (c.1951)","Alan Turing","2004","report","oxford.universitypressscholarship.com","oxford.universitypressscholarship.com/view/10.1093/oso/9780198250791.001.0001/isbn-9780198250791-book-part-18",0,"",""],["An Explainable Artificial Intelligence System for Small-unit Tactical Behavior","Michael van Lent and 2 others","2004","report","researchgate.net","www.researchgate.net/publication/221606722_An_Explainable_Artificial_Intelligence_System_for_Small-unit_Tactical_Behavior",0,"",""],["Designing the Whyline: A Debugging Interface for Asking Questions about Program Behavior","Andrew Ko and Brad A. Myers","2004","report","researchgate.net","www.researchgate.net/publication/221518887_Designing_the_Whyline_A_debugging_interface_for_asking_questions_about_program_behavior",0,"",""],["A Critical Look at Risk Assessments for Global Catastrophes","Adrian Kent","2004","report","onlinelibrary.wiley.com","onlinelibrary.wiley.com/doi/abs/10.1111/j.0272-4332.2004.00419.x",0,"",""],["All of Statistics, Chapters 1-12 or more (an easy-to-read overview of the field)","Larry Wasserman","2004","report","link.springer.com","link.springer.com/book/10.1007%2F978-0-387-21736-9",0,"",""],["Apprenticeship Learning via Inverse Reinforcement Learning","Pieter Abbeel and Andrew Ng","2004","report","ai.stanford.edu","ai.stanford.edu/~ang/papers/icml04-apprentice.pdf",0,"",""],["Beyond Normal Accidents and High Reliability Organizations: The Need for an Alternative Approach to Safety in Complex Systems","Karen Marais and 2 others","2004","report","researchgate.net","www.researchgate.net/publication/250212613_Beyond_Normal_Accidents_and_High_Reliability_Organizations_The_Need_for_an_Alternative_Approach_to_Safety_in_Complex_Systems",0,"",""],["Coherent Extrapolated Volition","Eliezer Yudkowsky","2004","report","intelligence.org","intelligence.org/files/CEV.pdf",0,"",""],["Diagnostic reasoning with A-Prolog","Marcello Balduccini and Michael Gelfond","2003","paper","TPLP Vol 3(4&5) (2003) 425-461","arxiv.org/abs/cs/0312040",0,"","agents"],["A logic for reasoning about upper probabilities","Joseph Y. Halpern and Riccardo Pucella","2003","paper","Journal of AI Research 17, 2001, pp. 57-81","arxiv.org/abs/cs/0307069",0,"",""],["Beslutstödssystemet Dezzy - en översikt","Ulla Bergsten and 2 others","2003","paper","in Dokumentation 7 juni av Seminarium och fackutst\\\"allning om\n  samband, sensorer och datorer f\\\"or ledningssystem till f\\\"orsvaret\n  (MILINF'89), pp. 07B2:19-31, Enk\\\"oping, June 1989, Telub AB, V\\\"axj\\\"o, 1989","arxiv.org/abs/cs/0305033",0,"",""],["A Framework for the Safety of Agent-Environment Systems","Ramesh Bharadwa j","2003","report","apps.dtic.mil","apps.dtic.mil/sti/pdfs/ADA465311.pdf",0,"","agents"],["A Brief History of Generative Models for Power Law and Lognormal Distributions","Michael Mitzenmacher","2003","report","stat.berkeley.edu","www.stat.berkeley.edu/~aldous/Networks/1089229510.pdf",0,"",""],["Astronomical Waste: The Opportunity Cost of Delayed Technological Development: Nick Bostrom","Nick Bostrom","2003","report","nickbostrom.com","nickbostrom.com/astronomical/waste.pdf",0,"",""],["Ethical Issues in Advanced Artificial Intelligence","Nick Bostrom","2003","report","taylorfrancis.com","www.taylorfrancis.com/books/9781000108934/chapters/10.4324/9781003074991-7",0,"",""],["Understanding Formal Methods, Chapters 1-10","Jean-François Monin","2003","report","researchgate.net","www.researchgate.net/publication/279352680_Understanding_Formal_Methods",0,"",""],["“Non-Player Character”","Eliezer S. Yudkowsky","2003","blog","yudkowsky.net","www.yudkowsky.net/other/fiction/npc",0,"",""],["The Complexity of Decentralized Control of Markov Decision Processes","Daniel S. Bernstein and 3 others","2002","report","pubsonline.informs.org","pubsonline.informs.org/doi/abs/10.1287/moor.27.4.819.297",0,"",""],["A Spectrum of Applications of Automated Reasoning","Larry Wos","2002","paper","arXiv preprint","arxiv.org/abs/cs/0205078",0,"","robustness"],["The Evolved Radio and its Implications for Modelling the Evolution of Novel Sensors","Jon Bird and Paul Layzell","2002","report","people.duke.edu","people.duke.edu/~ng46/topics/evolved-radio.pdf",0,"",""],["Representing and Aggregating Conflicting Beliefs","Pedrito Maynard-Reid II and Daniel Lehmann","2002","paper","Proceedings of the Seventh International Conference on Principles\n  of Knowledge Representation and Reasoning (KR 2000), April 2000, pp. 153-164","arxiv.org/abs/cs/0203013",0,"","agents"],["Anthropic bias: observation selection effects in science and philosophy","Nick Bostrom","2002","report","goodreads.com","www.goodreads.com/book/show/2002987.Anthropic_Bias",0,"",""],["Synchronization on small-world networks","H. Hong and 2 others","2001","paper","Phys. Rev. E 65, 026139 (2002)","arxiv.org/abs/cond-mat/0110359",0,"",""],["A Mathematical Introduction to Logic, Chapters 0-3 (alternative to Boolos and Burgess)","Herbert Enderton","2000","report","store.elsevier.com","store.elsevier.com/A-Mathematical-Introduction-to-Logic/Herbert-Enderton/isbn-9780122384523/",0,"",""],["Algorithmic Statistics","Péter Gács and 2 others","2000","paper","IEEE Transactions on Information Theory, Vol. 47, No. 6, September\n  2001, pp 2443-2463","arxiv.org/abs/math/0006233",0,"",""],["Algorithms for Inverse Reinforcement Learning","Andrew Ng and Stuart Russell","2000","report","ai.stanford.edu","ai.stanford.edu/~ang/papers/icml00-irl.pdf",0,"",""],["A Theory of Universal Artificial Intelligence based on Algorithmic Complexity","Marcus Hutter","2000","paper","arXiv preprint","arxiv.org/abs/cs/0004001",0,"","agents theory"],["Why the Future Doesn’t Need Us","Bill Joy","2000","report","wired.com","www.wired.com/2000/04/joy-2/",0,"",""],["Multi-Agent Only Knowing","Joseph Y. Halpern and Gerhard Lakemeyer","2000","paper","arXiv preprint","arxiv.org/abs/cs/0001015",0,"","agents"],["How Complex Systems Fail","Richard Cook","2000","report","how.complexsystems.fail","how.complexsystems.fail/",0,"",""],["Do the desires of rational agents converge?","D. Sobel","1999","report","academic.oup.com","academic.oup.com/analysis/article-lookup/doi/10.1093/analys/59.3.137",0,"","agents"],["Policy invariance under reward transformations: Theory and application to reward shaping","Andrew Ng and 2 others","1999","report","people.eecs.berkeley.edu","people.eecs.berkeley.edu/~pabbeel/cs287-fa09/readings/NgHaradaRussell-shaping-ICML1999.pdf",0,"","policy"],["Learning agents for uncertain environments","Stuart Russell","1998","report","portal.acm.org","portal.acm.org/citation.cfm?doid=279943.279964",0,"","agents"],["Long-Term Growth as a Sequence of Exponential Modes","Robin Hanson","1998","report","deliverypdf.ssrn.com","deliverypdf.ssrn.com/delivery.php?ID=778127031007078083000091124016094026052002093065027039103118108064007064092111076068110106060011059022008030089069103093111093122019094003010007028116115087119096101095030063101105000016081125114123084116006105103075120120108108067113103108011022009120&EXT=pdf&INDEX=TRUE",0,"",""],["Towards Flexible Teamwork","M. Tambe","1997","paper","Journal of Artificial Intelligence Research, Vol 7, (1997), 83-124","arxiv.org/abs/cs/9709101",0,"","agents"],["On the Interpretation of Decision Problems with Imperfect Recall","Michele Piccione and A. Rubinstein","1996","report","arielrubinstein.tau.ac.il","arielrubinstein.tau.ac.il/papers/53.pdf",0,"",""],["An Integrated Framework for Learning and Reasoning","C. G. Giraud-Carrier and T. R. Martinez","1995","paper","Journal of Artificial Intelligence Research, Vol 3, (1995),\n  147-185","arxiv.org/abs/cs/9508102",0,"",""],["Provably Bounded-Optimal Agents","S. J. Russell and D. Subramanian","1995","paper","Journal of Artificial Intelligence Research, Vol 2, (1995),\n  575-609","arxiv.org/abs/cs/9505103",0,"","agents"],["Provably Bounded-Optimal Agents","S. J. Russell and D. Subramanian","1995","report","jair.org","www.jair.org/index.php/jair/article/view/10134",0,"","agents"],["Artificial Intelligence: A Modern Approach","Stuart Russell and Peter Norvig","1994","report","goodreads.com","www.goodreads.com/book/show/27543.Artificial_Intelligence",0,"",""],["Linear Matrix Inequalities in System and Control Theory","Stephen Boyd and 3 others","1994","report","stanford.edu","stanford.edu/~boyd/lmibook/lmibook.pdf",0,"",""],["The Professional's Dilemma","Ben Hoffman","1990","report","mediangroup.org","mediangroup.org/docs/the_professionals_dilemma.pdf",0,"",""],["Incentive Compatibility and the Bargaining Problem","Roger B. Myerson","1979","report","jstor.org","www.jstor.org/stable/1912346",0,"",""],["Appendix I of Systemantics: How Systems Work and Especially How They Fail","John Gall","1977","report","drive.google.com","drive.google.com/file/d/1avoVTY8L3hpZi9fTxI_1mjjXC5JTz882/view?usp=sharing",0,"",""],["Machines and the Theory of Intelligence","Donald Michie","1973","report","nature.com","www.nature.com/articles/241507a0",0,"",""],["Incentives in Teams","Theodore Groves","1973","report","jstor.org","www.jstor.org/stable/1914085",0,"",""],["When is a Linear Control System Optimal","R.E. Kalman","1964","report","fluidsengineering.asmedigitalcollection.asme.org","fluidsengineering.asmedigitalcollection.asme.org/article.aspx?articleid=1431588",0,"",""],["The Last Question","Isaac Asimov","1956","report","users.ece.cmu.edu","users.ece.cmu.edu/~gamvrosi/thelastq.html",0,"",""],["Can digital machines think?","Alan Turing","1951","report","aperiodical.com","aperiodical.com/wp-content/uploads/2018/01/Turing-Can-Computers-Think.pdf",0,"",""]]}
