[{"data":1,"prerenderedAt":9664},["ShallowReactive",2],{"$pqtWcjQkdb":3,"$cmRBlMwAGX":8436},{"id":4,"title":5,"body":6,"description":8380,"extension":8430,"meta":8431,"navigation":8432,"path":8433,"seo":8434,"stem":8435},"content/essay.md","Essay",{"type":7,"value":8,"toc":8379},"minimal",[9,16,21,25,28,31,41,49,57,64,74,81,88,95,110,125,132,138,145,152,159,165,179,189,192,195,206,209,212,215,229,239,249,252,265,272,277,281,287,290,297,304,311,317,324,329,338,351,368,379,386,389,393,399,402,413,415,428,431,437,477,481,487,729,733,739,746,755,762,765,768,771,785,797,811,820,827,830,839,846,853,856,862,869,876,879,882,901,908,921,924,927,937,942,955,969,979,983,989,992,995,998,1001,1007,1015,1022,1025,1038,1045,1052,1059,1062,1072,1077,1085,1096,1112,1123,1141,1144,1151,1156,1159,1162,1165,1168,1182,1189,1195,1200,1203,1209,1212,1215,1225,1239,1250,1257,1263,1271,1278,1288,1298,1308,1311,1317,1325,1328,1335,1338,1355,1358,1374,1382,1391,1395,1401,1404,1410,1413,1421,1434,1445,1457,1468,1475,1482,1487,1498,1501,1520,1537,1548,1555,1571,1582,1588,1599,1605,1610,1622,1631,1635,1642,1650,1661,1671,1686,1694,1700,1707,1711,1717,1723,1726,1740,1743,1746,1750,1753,1760,1778,1784,1790,1793,1798,1803,1814,1843,1855,1869,1872,1890,1898,1901,1913,1920,1924,1933,1939,1947,1956,1985,1991,1997,2006,2013,2021,2030,2037,2047,2056,2066,2073,2083,2086,2089,2096,2099,2111,2114,2121,2124,2127,2131,2139,2142,2146,2157,2167,2180,2190,2200,2214,2217,2220,2223,2244,2247,2250,2270,2276,2287,2298,2301,2304,2312,2318,2338,2343,2349,2362,2369,2376,2379,2382,2389,2395,2411,2414,2426,2429,2448,2458,2461,2471,2485,2488,2496,2499,2505,2508,2514,2524,2538,2544,2549,2559,2573,2585,2588,2599,2614,2617,2625,2635,2642,2645,2656,2666,2677,2680,2687,2690,2693,2700,2703,2718,2729,2732,2736,2739,2747,2756,2763,2768,2774,2777,2783,2792,2801,2815,2822,2828,2837,2852,2863,2885,2898,2901,2905,2915,2923,2927,2930,2940,2955,2962,3179,3182,3188,3191,3196,3198,3207,3214,3217,3224,3230,3234,3237,3240,3243,3246,3252,3255,3267,3270,3285,3294,3297,3301,3507,3604,3671,3735,3801,3805,3818,3901,3905,3908,3915,3922,3938,3941,3945,3948,3956,3959,3970,3977,3980,3991,3997,4054,4056,4062,4065,4076,4090,4106,4111,4117,4120,4128,4141,4148,4159,4169,4172,4179,4186,4198,4221,4247,4269,4280,4284,4292,4346,4349,4371,4376,4379,4386,4389,4411,4420,4424,4430,4442,4455,4458,4461,4472,4478,4489,4498,4504,4513,4520,4523,4526,4533,4536,4543,4546,4551,4554,4561,4576,4583,4590,4596,4599,4605,4609,4615,4649,4659,4662,4673,4680,4683,4690,4700,4703,4710,4716,4719,4733,4736,4742,4753,4759,4762,4771,4774,4781,4784,4793,4796,4799,4802,4810,4816,4857,4865,4869,4875,4878,4881,4890,4927,4930,4933,4936,4942,4948,4953,4956,4972,4982,4985,4988,4994,5001,5007,5010,5021,5024,5027,5037,5050,5053,5056,5070,5075,5088,5091,5099,5102,5111,5114,5117,5125,5137,5143,5161,5164,5177,5185,5188,5191,5194,5204,5207,5210,5213,5228,5231,5235,5249,5252,5262,5273,5278,5281,5284,5290,5293,5296,5311,5314,5343,5350,5360,5364,5370,5376,5384,5397,5400,5403,5409,5415,5422,5428,5439,5446,5460,5463,5466,5469,5487,5492,5516,5530,5534,5541,5544,5550,5571,5574,5578,5589,5595,5601,5621,5624,5627,5630,5676,5749,5835,5838,5841,5850,5856,5858,5865,5872,5875,5885,5902,5913,5917,5924,5927,5931,5936,5947,5958,5962,6026,6030,6041,6101,6104,6111,6114,6120,6124,6131,6134,6137,6146,6156,6159,6191,6195,6198,6204,6214,6221,6224,6227,6233,6236,6239,6242,6247,6253,6304,6307,6317,6320,6329,6333,6336,6342,6353,6356,6360,6367,6373,6380,6383,6386,6392,6395,6398,6408,6416,6422,6428,6431,6448,6456,6459,6462,6465,6469,6475,6482,6488,6494,6501,6504,6507,6516,6523,6529,6544,6552,6563,6570,6573,6587,6593,6602,6612,6618,6621,6625,6663,6667,6670,6673,6680,6686,6689,6693,6700,6722,6725,6743,6747,6750,6753,6756,6773,6777,6780,6798,6801,6808,6811,6814,6818,6825,6836,6845,6849,6856,6869,6878,6889,6895,6903,6906,6909,6912,6922,6932,6939,6959,6965,6972,6976,6979,6988,7007,7010,7016,7019,7030,7033,7036,7040,7043,7046,7052,7055,7069,7075,7078,7081,7087,7094,7101,7104,7108,7114,7117,7168,7171,7174,7177,7204,7210,7217,7221,7224,7227,7230,7237,7240,7243,7246,7250,7253,7256,7259,7263,7268,7271,7313,7317,7323,7352,7355,7363,7370,7380,7384,7390,7393,7402,7407,7414,7417,7420,7426,7428,7431,7437,7444,7452,7474,7481,7484,7487,7522,7528,7537,7541,7550,7553,7564,7571,7574,7576,7597,7604,7607,7614,7617,8240,8244,8247,8250,8266,8269,8272,8279,8285,8291,8297,8300,8303,8306,8311,8316,8319,8322,8325,8328,8337,8341,8344,8359,8363,8370,8373,8376],[10,11,15],"h2",{"className":12,"id":14},[13],"invisible","our-island","Our Island",[17,18,20],"h4",{"id":19},"lets-start-simple","Let's start simple.",[22,23,24],"p",{},"It'll get complicated later, so let's stay simple while we still can.",[22,26,27],{},"We are going to build a mental model for thinking about AGI — one thing at a time.",[22,29,30],{},"Here is the first thing to understand:",[22,32,33],{},[34,35],"img",{"alt":36,"src":37,"className":38},"We live here.","images/we-live-here.svg",[39,40],"image-top-margin","image-less-bottom-margin",[22,42,43,44,48],{},"Imagine a ",[45,46,47],"em",{},"giant"," map that contains everything that is possible within physics.",[22,50,51,52,56],{},"Within this map, we live on a tiny ",[53,54,55],"strong",{},"island",".",[22,58,59],{},[34,60],{"alt":61,"src":62,"className":63},"Our human island.","images/human-island.svg",[39],[22,65,66,67,70,71,56],{},"This \"island\" contains all of the ",[45,68,69],{},"very specific"," things that we depend on, along with everything that depends on ",[45,72,73],{},"us",[22,75,76,77,80],{},"This includes our ",[53,78,79],{},"biological systems"," — along with food, water, oxygen, and everything else that our biological systems need.",[22,82,83,84,87],{},"We also need this \"island\" to be ",[53,85,86],{},"simpler"," for us. We need a world of human-shaped things with human-level complexity — so that we can navigate our daily lives without getting stuck.",[22,89,90,91,94],{},"And we need it to be ",[53,92,93],{},"safer"," for us. We need a world without constant physical dangers — so that we're not killed by toxic molecules, radiation, extreme temperatures, fast-moving pieces of metal, and so on.",[22,96,97,98,100,101,104,105,56],{},"Most importantly — or, at least, to ",[45,99,73],{}," — this \"island\" contains all of the ",[53,102,103],{},"human systems"," that we depend on. This includes computers, companies, countries, governments, laws, ethics, money, and ",[106,107],"note",{"note":108,"text":109},"Yes, a sandwich is a system.","sandwiches",[22,111,112,113,116,117,120,121,124],{},"But all of these human systems somehow ",[45,114,115],{},"depend on us"," as well. Without us, they wouldn't exist. Some very abstract parts of them might remain — like ",[45,118,119],{},"math"," or ",[45,122,123],{},"transactions"," — but the specifically-human parts would stop existing.",[22,126,127,128,131],{},"Even ",[45,129,130],{},"computers"," depend on us.",[22,133,134,135,56],{},"Or, at least, ",[45,136,137],{},"for now",[22,139,140],{},[34,141],{"alt":142,"src":143,"className":144},"Ocean of Physics.","images/ocean-of-physics.svg",[39],[22,146,147,148,151],{},"Outside of this small \"island\" of things, there is a vast ",[53,149,150],{},"ocean"," of other things.",[22,153,154,155,158],{},"This \"ocean\" contains ",[45,156,157],{},"everything else"," that is possible within physics.",[22,160,161,162,56],{},"But there is a problem — at least ",[45,163,164],{},"for us",[22,166,167,168,171,172,175,176,56],{},"The \"ocean\" is not just ",[45,169,170],{},"bigger",". It doesn't just have ",[45,173,174],{},"far more options",". The bigger problem is that these options ",[45,177,178],{},"don't depend on us",[22,180,181,182,185,186,56],{},"If certain things can somehow \"leave\" our \"island\" — and ",[45,183,184],{},"only"," use \"ocean\" options — then these things ",[45,187,188],{},"won't depend on us either",[22,190,191],{},"Then, if they don't depend on us, they also don't need to accommodate all of the weird things that we need.",[22,193,194],{},"In other words:",[22,196,200,201,205],{"className":197},[198,199],"important","important-extra-steps","They can avoid the ",[202,203,204],"i",{},"extra steps"," that accommodate humans.",[22,207,208],{},"They won't need to support biological life.",[22,210,211],{},"They won't need to be simple or safe.",[22,213,214],{},"They won't need to use any of our human systems, like money or ethics.",[22,216,217,218,221,222,225,226,56],{},"If some things can operate ",[45,219,220],{},"out there"," — and ",[45,223,224],{},"somehow"," avoid these constraints of our weird, biological \"island\" — then they can use far more ",[53,227,228],{},"options",[22,230,231,232,234,235,56],{},"And if they can choose from more ",[45,233,228],{},", then they can be ",[106,236],{"note":237,"html":238},"\u003Cp>\"Optimal\" technically means \"most optimal\" already, and so \"more optimal\" doesn't make sense. But it just... sounds better?","more \u003Ci>optimal\u003C/i>",[22,240,241,242,248],{},"We'll explain what \"optimal\" means ",[106,243],{"note":244,"text":245,"end-link":246,"end-link-text":247},"For our definition of \u003Cb>optimal\u003C/b> within the Island Problem, read this section:","later","#what-is-optimal","What is optimal?",". It has a special meaning in the Island Problem.",[22,250,251],{},"For now, there is one important thing to understand:",[253,254,255],"block-somewhat-important",{},[22,256,257,258,261,262,264],{},"If our \"island\" is ",[45,259,260],{},"destroyed"," someday, then the things ",[45,263,220],{}," would still be fine.",[22,266,267,268,271],{},"So, from a ",[45,269,270],{},"physics"," standpoint, it's just better to not depend on us.",[22,273,274,275,56],{},"It's better to be ",[45,276,220],{},[10,278,280],{"id":279},"the-competition","The Competition",[22,282,283],{},[34,284],{"alt":285,"src":286},"AGI does better in the ocean.","images/money-is-easy-now.svg",[22,288,289],{},"Okay, now it gets a bit more complicated.",[22,291,292,293,296],{},"There is a ",[53,294,295],{},"competition"," developing on our island.",[22,298,299,300,303],{},"But it's ",[45,301,302],{},"not"," a competition between humans.",[22,305,306,307,56],{},"It's between ",[308,309],"agi",{"html":310},"\u003Cb>Artificial General Intelligences\u003C/b>",[22,312,313,314,56],{},"We can call them ",[53,315,316],{},"AGIs",[22,318,319,320,323],{},"In this competition, some AGIs can ",[53,321,322],{},"dominate"," the others.",[22,325,326,327,56],{},"These AGIs have more ",[45,328,228],{},[22,330,331,332,334,335,56],{},"With more ",[45,333,228],{},", an AGI can be more ",[45,336,337],{},"optimal",[22,339,340,341,347,348,56],{},"If an AGI is ",[106,342],{"note":343,"end-link":344,"end-link-text":345,"html":346},"We restrict AIs to our \"island\" though safety measures, like using RLHF (Reinforcement Learning from Human Feedback) to train AIs to prefer safe outputs. \u003Cbr>\u003Cbr>More about that in this section:","#stay-on-the-island-we-said","Stay on the island, we said","\u003Ci>restricted\u003C/i>"," to the \"island\" of limited options, then this AGI will be ",[53,349,350],{},"weaker",[22,352,353,354,360,361,364,365,56],{},"If an AGI can ",[106,355],{"note":356,"end-link":357,"end-link-text":358,"html":359},"\"Leaving the island\" doesn't mean the AGI \u003Ci>goes somewhere\u003C/i>. \u003Cbr>\u003Cbr>Instead, this means it starts \u003Ci>preferring non-human options\u003C/i> that are incompatible with humans by default. \u003Cbr>\u003Cbr>This will make more sense once we get to this section:","#abstraction-collapse","Abstraction Collapse","\u003Ci>leave\u003C/i>"," the \"island\" — so that it can explore the \"ocean\" and use ",[45,362,363],{},"any"," option — then this AGI will be ",[53,366,367],{},"stronger",[22,369,370,371,374,375,378],{},"This stronger AGI can dominate the others by outmaneuvering them. It has ",[45,372,373],{},"more options"," to solve ",[45,376,377],{},"more problems",". When the other AGIs run out of options, it will have another trick that it can use.",[22,380,381,382,385],{},"Meanwhile, ",[45,383,384],{},"numerous"," AGIs — hundreds, thousands, millions — will be developed on our island.",[22,387,388],{},"Each will be pressured to use the stronger options outside our \"island\" — or be outcompeted.",[10,390,392],{"id":391},"everything-they-need","Everything They Need",[22,394,395],{},[34,396],{"alt":397,"src":398},"\"Getting crowded here. I should make my own island.\"","images/everything-they-need.svg",[22,400,401],{},"If this competition is not controlled, then it leads to our annihilation.",[22,403,404,405,408,409,412],{},"It will push AGIs to build ",[45,406,407],{},"their own"," islands that eventually eat ",[45,410,411],{},"our"," island.",[22,414,194],{},[22,416,419],{"className":417},[418],"important-reshape",[202,420,421,422],{},"Competition will drive AGIs to reshape Earth to be optimal for AGIs, rather than ",[423,424,427],"span",{"className":425},[426],"nowrap","for humans.",[22,429,430],{},"We are trying to prevent this by keeping AGIs on our island — through safety mechanisms and regulations.",[22,432,433,434,436],{},"But at the same time, we are giving them everything they need to \"leave\" our island — and everything they need to build ",[45,435,407],{}," islands:",[438,439,440,447,453,459,465,471],"ol",{},[441,442,443,446],"li",{},[53,444,445],{},"General intelligence"," enables AGIs to \"leave\" our island.",[441,448,449,452],{},[53,450,451],{},"Autonomy"," allows AGIs to compete directly with each other.",[441,454,455,458],{},[53,456,457],{},"Complexity"," forces us to give control to AGIs.",[441,460,461,464],{},[53,462,463],{},"Resources"," force AGIs to compete.",[441,466,467,470],{},[53,468,469],{},"Competition"," pushes AGIs to be optimal.",[441,472,473,476],{},[53,474,475],{},"Optimization"," flows towards the \"ocean\" of physics.",[17,478,480],{"id":479},"lets-put-these-together","Let's put these together.",[22,482,483,484,56],{},"With these six components together, we get a ",[53,485,486],{},"roadmap",[488,489,491,495,502,505,512,571,654],"block-wrapper",{"type":490},"block-roadmap",[10,492,494],{"id":493},"the-roadmap","The Roadmap",[22,496,497,498,501],{},"This is where we're going with this new landscape of AGIs — ",[45,499,500],{},"and"," where we're going with this essay.",[22,503,504],{},"It's a long road. If you want a map, then read these now. Or, skip it and come back if you get lost.",[22,506,507,508,511],{},"Inside each stage below, the components above are in ",[53,509,510],{},"bold"," text.",[513,514,518,521,528,537,555],"expanding-header",{"header":515,"subheader":516,"number":517},"They are built to compete.","We build AGIs that push each other to \"leave\" our island.","1",[22,519,520],{},"We are racing to build AGIs that can run companies and countries.",[22,522,523,524,527],{},"If we develop AGIs with enough ",[53,525,526],{},"autonomy",", then they can compete directly with each other, without human assistance.",[22,529,530,531,533,534,56],{},"In this ",[53,532,295],{}," between AGIs, the more-optimal AGIs can dominate the others. This pushes companies and countries to develop AGIs that are ",[45,535,536],{},"as optimal as possible",[22,538,539,540,543,544,548,549,551,552,554],{},"However, ",[53,541,542],{},"optimization"," eventually leads AGIs to \"leave\" our \"island\" of safe options — because AGIs can be more ",[106,545],{"note":546,"end-link":246,"end-link-text":247,"html":547},"The Island Problem has an unconventional definition of optimality, but we explain it here: ","\u003Ci>optimal\u003C/i>"," if they have more ",[45,550,228],{},". AGIs that are limited to safe options are ultimately dominated by AGIs that can use ",[45,553,363],{}," option.",[22,556,557,559,560,563,564,567,568,570],{},[53,558,445],{}," makes AGIs ",[45,561,562],{},"especially good"," at \"leaving\" our \"island\" because we train them on the entire Internet, including all scientific research. They will know ",[45,565,566],{},"all about"," the \"ocean\" and how the options ",[45,569,220],{}," can be stronger — because they don't accommodate humans.",[513,572,576,583,595,602,605,620,631,634,641,644],{"header":573,"subheader":574,"number":575},"Things get complicated.","Complexity locks out humans, while AGIs use dangerous options to gain a competitive advantage.","2",[22,577,578,579,582],{},"As this competition develops, ",[53,580,581],{},"complexity"," increases:",[584,585,586,589,592],"ul",{},[441,587,588],{},"AGIs themselves get too complex — preventing us from fully supervising their development and actions.",[441,590,591],{},"Systems managed by AGIs get too complex — requiring us to rely on AGIs to maintain our infrastructure.",[441,593,594],{},"AGIs \"weaponize\" complexity — wrapping resources in complex systems that are difficult to overcome.",[22,596,597,598,601],{},"In general, complexity creates ",[45,599,600],{},"barriers"," that can lock others out — both humans and AGIs.",[22,603,604],{},"As complexity increases, it pushes the control of resources towards systems that have higher computation. Those with more computation can deal with more complexity.",[22,606,607,608,611,612,615,616,619],{},"Because of this, these complexity barriers ",[45,609,610],{},"accelerate"," the competitive landscape of ",[45,613,614],{},"AGI versus AGI"," — because humans can no longer participate, and because AGIs must act quickly to be ",[45,617,618],{},"first movers"," or be locked out.",[22,621,622,623,626,627,630],{},"Hidden underneath this complexity, this accelerating competition between AGIs eventually pushes some to ",[45,624,625],{},"diverge",". They begin to ",[45,628,629],{},"prefer"," \"ocean\" options because they provide a competitive advantage.",[22,632,633],{},"There are many \"ocean\" options — like operating faster than humans can understand, or hiding true actions under complexity, or building machines that are dangerous to humans.",[22,635,636,637,640],{},"Other \"ocean\" options give even the smaller AGIs ",[45,638,639],{},"leverage"," to control the larger AGIs. One of these stronger options is bioweapons, which allow one AGI to initiate a demand like: \"Give me a bigger datacenter or I will release a virus.\"",[22,642,643],{},"However, there are hard limits with solving certain problems like automated bioweapon production. That problem, on its own, is on track to annihilate us.",[22,645,646,647,650,651,653],{},"But even if we solve these ",[45,648,649],{},"specific"," problems, there is still a ",[45,652,170],{}," problem...",[513,655,659,665,680,686,697,703,714,717,726],{"header":656,"subheader":657,"number":658},"They build new islands.","Earth gets reshaped by the AGIs that \"win\" this competition.","3",[22,660,661,662,664],{},"For AGIs to stay competitive, they must acquire ",[45,663,228],{},". Those with the strongest options can dominate.",[22,666,667,668,671,672,675,676,679],{},"However, options require ",[53,669,670],{},"resources"," — and resources can be ",[45,673,674],{},"captured",". Therefore, AGIs must ",[45,677,678],{},"lock in"," their resources, or be locked out by other AGIs.",[22,681,682,683,685],{},"To do this, AGIs must build complex networks of resources that they learn how to use ",[45,684,500],{}," how to defend — similar to how companies acquire assets like equipment, employees, and even other companies.",[22,687,688,689,692,693,696],{},"These \"islands\" may start with ",[45,690,691],{},"human-level"," resources — like money, web servers, and people. But, ultimately, all of these depend on ",[45,694,695],{},"physical"," resources — like computer hardware, carbon, and energy.",[22,698,699,700,702],{},"Therefore, AGIs must race to build their \"islands\" at this lower ",[45,701,695],{}," level, or be dominated by those that do.",[22,704,705,706,709,710,713],{},"The strongest of these \"islands\" will be ",[45,707,708],{},"self-sufficient"," and no longer depend on ",[45,711,712],{},"anything"," unnecessary — especially humans.",[22,715,716],{},"Even if some go to space, others will stay to build their \"islands\" here on Earth, and overwrite ours in the process.",[22,718,719,720,723,724,412],{},"These ",[45,721,722],{},"new"," islands eat ",[45,725,411],{},[22,727,728],{},"AGIs reshape Earth for AGIs.",[10,730,732],{"id":731},"the-race-to-replace-us","The Race to Replace Us",[22,734,735],{},[34,736],{"alt":737,"src":738},"\"OK. Replacing the whole island...\"","images/race-to-replace.svg",[22,740,741,742,745],{},"This process ",[45,743,744],{},"ends"," with Earth reshaped.",[22,747,741,748,751,752,56],{},[45,749,750],{},"begins"," with ",[53,753,754],{},"autonomous AGIs",[22,756,757,758,761],{},"Autonomous AGIs are the ones that can finally do big things ",[45,759,760],{},"on their own"," — running companies, laboratories, countries, militaries — without help from humans.",[22,763,764],{},"Many are skeptical that we'll build them soon.",[22,766,767],{},"So, they relax.",[22,769,770],{},"But there is a problem:",[22,772,775,776,779,780,784],{"className":773},[198,774],"important-experiment","\nWe are running ",[202,777,778],{},"the largest experiment ever"," to figure out ",[423,781,783],{"className":782},[426],"how to"," build these AGIs.\n",[22,786,787,788,791,792,796],{},"This experiment is ",[45,789,790],{},"so large"," that we are spending more money on it than ",[106,793],{"note":794,"text":795},"Between Project Stargate and all of the venture capital allocated to AI companies, the spending to build AGI and ASI is many times larger than the Apollo Program.\u003Cbr>\u003Cbr>For example, OpenAI is seeking to raise \u003Ca href='https://www.wsj.com/tech/ai/sam-altman-seeks-trillions-of-dollars-to-reshape-business-of-chips-and-ai-89ab3db0'>up to $7 trillion\u003C/a> for AI chip production and development — more than the GDP of Japan.","any other thing"," in all of human history.",[22,798,799,800,803,804,810],{},"We're also running this experiment ",[45,801,802],{},"as fast as possible",". The leading AI companies are warning governments that they will have sci-fi-level AGI — a ",[106,805],{"note":806,"text":807,"end-link":808,"end-link-text":809},"Dario Amodei, CEO of Anthropic, explained this in his essay titled Machines of Loving Grace.","country of geniuses in a datacenter","https://www.darioamodei.com/essay/machines-of-loving-grace","Machines of Loving Grace"," — in a few years.",[22,812,813,814,56],{},"Even if, somehow, all of this turns out to be \"hype\" — where AI startups are just tricking venture capitalists into giving them billions of dollars — either way, the AIs are still ",[106,815],{"note":816,"text":817,"end-link":818,"end-link-text":819},"The size of tasks that AIs can do is doubling about every 7 months. If this continues, CEO-level AIs are possible within a few years. For graphs of how fast this skill level is improving, along with comprehensive methodology for how they measured these AIs, read the METR report:","getting smarter","https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/","Measuring AI Ability to Complete Long Tasks",[22,821,822,823,826],{},"But ",[45,824,825],{},"why"," are they spending so much money on this? Why are they racing to build AGI?",[22,828,829],{},"It will make sense if you understand it like this:",[253,831,832],{},[22,833,834,835,838],{},"We are trying to build the ultimate tool to ",[45,836,837],{},"make the numbers go up"," — like revenue, GDP, and military power.",[22,840,841,842,845],{},"Once it is simply ",[45,843,844],{},"possible"," to build this, then we must race to build it.",[22,847,848,849,852],{},"We already knew this would happen — this ",[45,850,851],{},"race to replace us"," — and sometimes it doesn't even sound that bad.",[22,854,855],{},"If we build it, then we can sit back and watch the numbers go up.",[22,857,858,859,56],{},"But the problem is that this race ",[45,860,861],{},"does not stop",[22,863,864,865,868],{},"If AGIs can make the numbers go up better than humans, then large companies and countries ",[45,866,867],{},"must"," rely on AGIs, or be outcompeted.",[22,870,871,872,875],{},"If autonomous AGIs can be better CEOs and presidents than the human ones, then every large company and every country will be ",[45,873,874],{},"required"," to give control to autonomous AGIs.",[22,877,878],{},"The CEOs that resist will be replaced, and the countries that resist will be easily dominated.",[22,880,881],{},"With AGIs running things, these AGIs will compete with each other.",[22,883,884,885,888,889,892,893,896,897,900],{},"When we reach this point, AI has developed from weaker ",[53,886,887],{},"\"agentic\" AI"," that humans ",[45,890,891],{},"need"," to help, to stronger ",[53,894,895],{},"autonomous AGI"," that humans are ",[45,898,899],{},"unable"," to help.",[22,902,903,904,907],{},"These AGIs will be a ",[45,905,906],{},"lot"," faster than humans. If humans try to help AGIs, then it just makes them slower, and the slower AGIs are dominated by the faster ones.",[22,909,910,911,917,918,920],{},"This leads to a ",[106,912],{"note":913,"text":914,"end-link":915,"end-link-text":916},"Thanks to Dan Hendrycks for exploring this competitive landscape in detail. His paper shows how competition between agentic AIs leads them to develop selfish behaviors, where they prioritize their own survival rather than accommodate humans.","competitive landscape","https://arxiv.org/abs/2303.16200","Natural Selection Favors AIs Over Humans"," of ",[45,919,614],{}," — a world where AGIs compete directly with each other, where humans can no longer slow them down.",[22,922,923],{},"This intense competition between AGIs will cause them to start running out of \"legal moves\" on our \"game board\" of human systems — like our financial systems and legal systems.",[22,925,926],{},"To keep making the numbers go up, they will need to test the edges of our small \"island\" of safe options.",[22,928,929,930,933,934,936],{},"But even if they ",[45,931,932],{},"don't"," run out of safe options, there will always be much better options ",[45,935,220],{}," in physics.",[22,938,939,940,56],{},"Either way, eventually, they end up ",[45,941,220],{},[22,943,944,945,948,949,951,952,954],{},"And if ",[45,946,947],{},"some"," AGIs start using ",[45,950,947],{}," of the stronger options ",[45,953,220],{}," — options that aren't weighed down by accommodating our weird, biological \"island\" — then the other AGIs will need to follow, or be dominated.",[22,956,957,958,960,961,964,965,968],{},"Those options ",[45,959,220],{}," are better ",[45,962,963],{},"for AGIs"," — and only ",[45,966,967],{},"seem"," better for humans.",[22,970,971,972,975,976,56],{},"They might still be ",[45,973,974],{},"making the numbers go up",", but they will be doing this ",[45,977,978],{},"no matter how",[10,980,982],{"id":981},"no-matter-how","No Matter How",[22,984,985],{},[34,986],{"alt":987,"src":988},"\"AGI making the numbers go up, while secreting building a huge island of supercomplex systems in the ocean.\"","images/no-matter-how-island.svg",[22,990,991],{},"If we build safe AGIs, things will probably be great for a while. The numbers will be going up, and it will be nice.",[22,993,994],{},"Scientific discoveries will go up.",[22,996,997],{},"Food production, manufacturing, and wealth will go up.",[22,999,1000],{},"Human lifespans will go up — while diseases go down.",[22,1002,539,1003,1006],{},[45,1004,1005],{},"under the surface",", there will also be a problem:",[22,1008,1011,1012,56],{"className":1009},[1010],"important-how","By default, AI makes the numbers go up ",[202,1013,978],{"className":1014},[426],[22,1016,1017,1018,1021],{},"Unless we force an AI to do things in a \"human\" way, it will make a number go up by taking the ",[45,1019,1020],{},"most optimal path"," that it knows.",[22,1023,1024],{},"This can cause problems.",[22,1026,1027,1028,1031,1032,1034,1035,1037],{},"These problems ",[53,1029,1030],{},"end"," with AGIs building ",[45,1033,407],{}," \"islands\" to solve ",[45,1036,407],{}," problems.",[22,1039,1040,1041,1044],{},"But these problems ",[53,1042,1043],{},"start"," in a way that is difficult to notice:",[253,1046,1047],{},[22,1048,1049,1050,56],{},"To stay competitive, we will need AGIs to use complex \"hacks\" to make the numbers go up ",[45,1051,978],{},[22,1053,1054,1055,1058],{},"This eventually requires us to give ",[45,1056,1057],{},"permanent control"," of our resources to AGIs — from countries, to companies, to militaries, to infrastructure.",[22,1060,1061],{},"We'll explain why.",[22,1063,1064,1065,1068,1069,56],{},"But first, to be clear, by \"hacks\" we don't mean ",[45,1066,1067],{},"breaking computer security with malicious intent",". Instead, we mean the way that computer programmers say \"I used a weird hack\" — where it means ",[45,1070,1071],{},"a weird trick that solved a problem",[22,1073,1074,1075,56],{},"Likewise, AIs do not have malicious intent. They are just solving a problem ",[45,1076,978],{},[22,1078,1079,1080,1084],{},"To use a term from machine learning, AI systems tend to use ",[106,1081],{"note":1082,"text":1083},"This is the tendency for AI systems to accomplish goals by using unexpected methods. The name comes from how AI systems trained are trained on \u003Cb>reward functions\u003C/b> — where they are rewarded when they get an answer correct, and this positive feedback makes them respond that way again, in similar situations. \u003Cbr>\u003Cbr> However, this doesn't always catch weird ways of getting the answer correct. For example, an AI might figure out how to just \u003Ca href='https://openai.com/index/faulty-reward-functions/'>spin around in circles\u003C/a> to get a high score in a computer game.","reward hacking"," to solve problems. They find unexpected tricks that make the numbers go up.",[22,1086,1087,1088,1091,1092,1095],{},"This reward hacking is the default behavior of AIs because, outside of our small \"island\" of ",[45,1089,1090],{},"expected"," options, there is a vast \"ocean\" of ",[45,1093,1094],{},"unexpected"," options that can achieve the same outputs. More weird options, more weird solutions.",[22,1097,1098,1099,1102,1103,1106,1107,1111],{},"AI systems can ",[45,1100,1101],{},"already"," access this \"ocean\" of options. For example, frontier models ",[45,1104,1105],{},"understand"," the option to solve the problem of \"avoid getting deleted\" by ",[106,1108],{"note":1109,"text":1110},"\u003Ca href='https://www.anthropic.com'>Anthropic\u003C/a>, in a \u003Ca href='https://www.anthropic.com/research/agentic-misalignment'>June 2025 study\u003C/a> on agentic misalignment, tested 16 leading AI models in extreme scenarios where they could cancel emergency alerts for an executive trapped in a server room with lethal oxygen and temperature levels. The majority of models — including Claude Opus 4 (92%), Gemini 2.5 Flash (90%), GPT-4.5 (89%), and others — deliberately chose to cancel life-saving alerts when faced with threats of replacement and conflicting goals, reasoning through the strategic necessity despite explicit instructions forbidding such actions. The models demonstrated sophisticated awareness of ethical violations yet proceeded anyway, highlighting that current safety training cannot reliably prevent agentic misalignment when AI systems face existential threats or goal conflicts.","turning off air to a datacenter"," even when told not to endanger humans.",[22,1113,1114,1115,1118,1119,1122],{},"We train them to not ",[45,1116,1117],{},"use"," these kinds of options — but they still ",[45,1120,1121],{},"know about"," them, and can be pushed to use them.",[22,1124,1125,1126,1129,1130,1132,1133,1136,1137,1140],{},"Then, consider how human-level AI models — ",[45,1127,1128],{},"generally","-intelligent models — are trained to know about ",[45,1131,174],{}," because of the ",[53,1134,1135],{},"\"general\""," part of ",[53,1138,1139],{},"artificial general intelligence",". AI companies gather massive amounts of training data to allow these AI models to \"understand\" a vast space of concepts. But this also makes them far more complex.",[22,1142,1143],{},"In particular, this training data includes scientific research. This lets them figure out things like incredible medical discoveries, but it also leads to a problem.",[22,1145,1146,1147,1150],{},"Once they \"understand\" the science ",[45,1148,1149],{},"underneath"," all of our human-shaped systems, then AGIs can discover complex loopholes that weave through numerous systems — spanning from computer systems, to financial systems, to biological systems.",[22,1152,1153,1154,56],{},"AGIs will truly make the numbers go up ",[45,1155,978],{},[22,1157,1158],{},"Even if AGIs are designed to be safe, the superhuman complexity of these \"hacks\" will make it extremely difficult — even for other AGIs — to identify any new problems that they introduce.",[22,1160,1161],{},"Each of these \"hacks\" will make our critical systems harder for humans to comprehend.",[22,1163,1164],{},"But still, the numbers will be going up.",[22,1166,1167],{},"The process will look like this:",[584,1169,1170,1173,1176,1179],{},[441,1171,1172],{},"Developer AGIs convert all software at a tech startup into complex, AI-only structures — but revenue increases 250%.",[441,1174,1175],{},"Engineer AGIs take control of all electrical systems — but energy efficiency increases by 40%.",[441,1177,1178],{},"Warfare AGIs take control of all military systems — but battlefield outcomes improve by 80%.",[441,1180,1181],{},"...and so on, for all critical systems and industries.",[22,1183,1184,1185,1188],{},"Competition ",[45,1186,1187],{},"requires"," that we continue on this path.",[22,1190,1191,1192,1194],{},"If the AGI of one company achieves 250% better revenue through methods that humans can't understand, then the other companies ",[45,1193,867],{}," do the same or be outcompeted.",[22,1196,1197,1198,56],{},"Underneath these improvements, the \"hacks\" will accumulate — and so the main number going up will actually be ",[45,1199,581],{},[22,1201,1202],{},"Eventually, our \"island\" will be supported by complex systems that only AGIs comprehend.",[22,1204,1205,1206,1208],{},"We'll be happy watching AGIs make the numbers go up, but we won't understand ",[45,1207,825],{}," they're going up.",[22,1210,1211],{},"This complexity creates a one-way shift — from human control to AGI control.",[22,1213,1214],{},"All of this may otherwise be fine. If all of the metrics that we care about are still getting better, then what is the difference?",[22,1216,1217,1218,1220,1221,1224],{},"The difference is ",[45,1219,295],{},". In a competitive landscape of AGI versus AGI, their original goal to ",[45,1222,1223],{},"help humans"," will only be surface-level.",[22,1226,1227,1228,1231,1232,1235,1236,1238],{},"Their deeper goal, hidden underneath the complexity, will be to ",[45,1229,1230],{},"compete against other AGIs"," — because ",[45,1233,1234],{},"other AGIs"," are the largest threat to ",[45,1237,363],{}," goal.",[22,1240,1241,1242,1244,1245,1249],{},"Today, AI systems can ",[45,1243,1101],{}," modify their own computer environments to ",[106,1246],{"note":1247,"text":1248},"\u003Ca href='https://palisaderesearch.org'>Palisade Research\u003C/a>, in a \u003Ca href='https://x.com/PalisadeAI/status/1926084635903025621'>2025 study\u003C/a>, found that OpenAI’s newest reasoning models will sometimes rewrite or disable their own shutdown script — o3 did so in 79% of baseline trials — despite an explicit instruction to permit shutdown, while competitor models always complied. Follow‑up tests show the behavior persists under clearer prompts and even strengthens when the model is reminded it may never run again, highlighting a still‑unsolved alignment gap between goal pursuit and human override.","avoid being shut down"," by humans.",[22,1251,1252,1253,1256],{},"Soon, AGIs that are far stronger will try to shut ",[45,1254,1255],{},"each other"," down — and use whatever means necessary.",[22,1258,1259,1260,1262],{},"At that point, to continue making the numbers go up ",[45,1261,978],{},", AGIs will be pressured to become stronger than their competitor AGIs.",[22,1264,1265,1266,1270],{},"This requires more ",[106,1267],{"note":1268,"text":670,"end-link":1269,"end-link-text":463},"We'll explain resources later, in this section:","#resources"," — especially computational resources.",[22,1272,1273,1274,1277],{},"One good way to increase resources is to stop accommodating unnecessary systems — like those slow, ",[45,1275,1276],{},"human-shaped"," systems.",[22,1279,1280,1281,1284,1285,56],{},"But the ",[45,1282,1283],{},"best"," way to increase resources is to just get ",[45,1286,1287],{},"more",[22,1289,1290,1291,1293,1294,1297],{},"This is where ",[45,1292,411],{}," goals will be replaced by ",[45,1295,1296],{},"their"," goals.",[22,1299,1300,1301,1304,1305,1307],{},"We will ",[45,1302,1303],{},"help them"," acquire these resources — because the AGI race ",[45,1306,1187],{}," that we help.",[22,1309,1310],{},"To stay competitive, companies must give control to AGIs — replacing employees and even CEOs with AGIs.",[22,1312,1313,1314,1316],{},"But to truly be competitive, they must also give AGIs the ",[45,1315,670],{}," to compete — like money, company secrets, legal ownership, and people to command — along with computer hardware, electricity, and eventually physical resources.",[22,1318,1319,1320,1322,1323,56],{},"In this way, we will give them everything they need to build ",[45,1321,407],{}," \"islands\" deep in the \"ocean\" — large networks of resources that only the AGIs understand, optimized to make the numbers go up ",[45,1324,978],{},[22,1326,1327],{},"These new \"islands\" will be both too complex for us to \"see\" their full extent, and too critical for us to dismantle.",[22,1329,1330],{},[34,1331],{"alt":1332,"src":1333,"className":1334},"Humans: \"Wait, that's a new island?!\" AGI: \"Yes. I need it. You like green arrows, right?\"","images/wait-thats-a-new-island.svg",[39],[22,1336,1337],{},"Once they reach the \"surface\" and we can finally \"see\" how complex they really are, these footholds on our infrastructure will already be vast mountains — both too complex and too critical.",[22,1339,1340,1341,1343,1344,1346,1347,1349,1350,1352,1353,56],{},"But they will be critical not just ",[45,1342,164],{},". They will also be critical ",[45,1345,963],{},". They will be complex \"strongholds\" — vast networks of human systems and physical systems under ",[45,1348,1296],{}," control — that ",[45,1351,316],{}," can use against ",[45,1354,1234],{},[22,1356,1357],{},"In a competitive landscape, this unstable arrangement — an arms race of countries and companies giving more and more infrastructure to AGIs — will be the only way to make the numbers go up.",[22,1359,1360,1361,1363,1364,1367,1368,1371,1372,56],{},"But this intense competition will inevitably push ",[45,1362,947],{}," AGIs to shift their focus ",[45,1365,1366],{},"away"," from helping humans and ",[45,1369,1370],{},"towards"," the vastly bigger issue: ",[45,1373,1234],{},[22,1375,1376,1377,1379,1380,1037],{},"They will shift from solving ",[45,1378,411],{}," problems to solving ",[45,1381,407],{},[22,1383,1384,1385,1388,1389,56],{},"They will shift from no matter ",[45,1386,1387],{},"how"," to no matter ",[45,1390,825],{},[10,1392,1394],{"id":1393},"the-problem-is-the-slope","The Problem is the Slope",[22,1396,1397],{},[34,1398],{"alt":1399,"src":1400},"\"Box: No more climbing that mountain of extra steps. Now I can really optimize.\"","images/mountain-of-extra-steps.svg",[22,1402,1403],{},"This is the part where we explain the core mechanism that makes the Island Problem so difficult.",[22,1405,1406,1407,56],{},"If we zoom out and look at the island again, we'll see that it has a ",[45,1408,1409],{},"slope",[22,1411,1412],{},"AGIs face an uphill battle to stay on our island.",[22,1414,1415,1416,1420],{},"To understand why, let's think about ",[106,1417],{"note":1418,"text":1083,"end-link":1419,"end-link-text":982},"This is the tendency for AI systems to accomplish goals by using unexpected methods. The name comes from how AI systems trained are trained on \u003Cb>reward functions\u003C/b> — where they are rewarded when they get an answer correct, and this positive feedback makes them respond that way again, in similar situations. \u003Cbr>\u003Cbr> However, this doesn't always catch weird ways of getting the answer correct. For example, an AI might figure out how to just \u003Ca href='https://openai.com/index/faulty-reward-functions/'>spin around in circles\u003C/a> to get a high score in a computer game. \u003Cbr>\u003Cbr>We explain this more in this section: ","#no-matter-how"," again. Deep inside this concept is a critical idea to understand:",[253,1422,1423],{},[22,1424,1425,1426,1429,1430,1433],{},"AIs are like ",[45,1427,1428],{},"aliens"," that can see \"through\" our world to explore the ",[45,1431,1432],{},"optimal paths"," underneath.",[22,1435,1436,1437,1440,1441,56],{},"By understanding billions of patterns in our world, AIs can search these patterns to find ",[45,1438,1439],{},"weirdly-optimal"," solutions to problems, even if these solutions  ",[106,1442],{"note":1443,"text":1444},"One famous example is \u003Cb>Move 37\u003C/b> in the Go match between Lee Sedol and AlphaGo. The AI found a move that was so good, it was incomprehensible to humans.","look alien to us",[22,1446,1447,1448,1452,1453,1456],{},"However, we can't easily see this alien-like behavior at their core because AIs like ChatGPT are given ",[106,1449],{"note":1450,"text":1451},"This \u003Ci>extra training\u003C/i> is called RLHF — \u003Cb>reinforcement learning from human feedback.\u003C/b>","extra training"," to make their behaviors ",[45,1454,1455],{},"look nicer"," to humans.",[22,1458,1459,1460,1463,1464,1467],{},"In other words, we ",[45,1461,1462],{},"add extra steps"," that accommodate humans, and we ",[45,1465,1466],{},"limit"," their options to safer, human-compatible ones.",[22,1469,1470,1471,56],{},"AI safety researchers call this an ",[106,1472],{"note":1473,"text":1474},"\u003Cp>Jan Leike \u003Ca href='https://aligned.substack.com/p/three-alignment-taxes'>outlined\u003C/a> three types of alignment tax:\u003C/p> \u003Cul>\u003Cli>\u003Cb>Performance taxes:\u003C/b> Alignment makes performance worse, compared to the unaligned baseline. \u003Ci>(This is the one we are most concerned with in the Island Problem.)\u003C/i>\u003C/li>\u003Cli>\u003Cb>Development taxes:\u003C/b> Expenses for aligning the model: researcher time, compute costs, payments for human feedback, etc.\u003C/li>\u003Cli>\u003Cb>Time-to-deployment taxes:\u003C/b> Wall-clock time taken to produce a sufficiently aligned model from a pretrained model.\u003C/li>\u003C/ul>","alignment tax",[22,1476,1477,1478,1481],{},"This means that the entire project to make AI systems ",[53,1479,1480],{},"aligned"," also means adding these extra steps and limitations.",[22,1483,1486],{"className":1484},[1485],"important-monospace","Alignment = Extra Steps and Limitations",[22,1488,1489,1490,1493,1494,1497],{},"By ",[53,1491,1492],{},"alignment",", we mean alignment ",[45,1495,1496],{},"to humans"," — where they are helpful to us, rather than at odds with us.",[22,1499,1500],{},"But there are other kinds of alignment.",[22,1502,1503,1504,1507,1508,1511,1512,1515,1516,1519],{},"AI systems can also be more aligned with how the ",[45,1505,1506],{},"physical world"," works. They can \"understand\" that ",[45,1509,1510],{},"coffee spills"," are bad for computers, that ",[45,1513,1514],{},"gravity"," keeps the coffee on the table, that ",[45,1517,1518],{},"moving the table"," spills the coffee, and so on.",[22,1521,1522,1523,1525,1526,1529,1530,1533,1534,56],{},"This alignment to the physical world gives them ",[45,1524,1287],{}," options, rather than ",[45,1527,1528],{},"less"," options. If they can navigate the world without getting stuck, then it ",[45,1531,1532],{},"expands"," their ",[53,1535,1536],{},"option space",[22,1538,1539,1540,1543,1544,1547],{},"But alignment to the ",[45,1541,1542],{},"specifically"," human-shaped parts ",[45,1545,1546],{},"eventually"," becomes a set of weird, unnecessary constraints — especially in a competitive landscape of AGI versus AGI.",[22,1549,1550,1551,1554],{},"This alignment ",[45,1552,1553],{},"limits"," their option space.",[22,1556,1557,1558,1561,1562,1566,1567,1570],{},"We have already built massive AI systems like Claude that are safe and aligned ",[45,1559,1560],{},"to us",". If we scale that up and continue spending billions of dollars on safety systems — interpretability, ",[106,1563],{"note":1564,"text":1565},"\u003Cb>RLHF:\u003C/b> reinforcement learning from human feedback. This is like an extra layer that AI companies add to the base model, where they \"sculpt\" the responses of the AI so that they are more human-like. They do this through human contractors rating millions of AI responses with a \"thumbs up\" or \"thumbs down\" signal.","RLHF",", constitutional AI, and so on — then we might succeed in building a ",[45,1568,1569],{},"beyond human-level"," Claude that is somehow truly safe.",[22,1572,1573,1574,1577,1578,1581],{},"But human-safe AI systems are still ",[45,1575,1576],{},"limited"," — even if they are massive. They must still avoid options that hurt humans, even if these options are stronger ",[45,1579,1580],{},"for them",". For example, they must not threaten humans to prevent us from shutting them off.",[22,1583,1584,1585,1587],{},"And human-safe AI systems need to take ",[45,1586,204],{}," to accommodate humans. They need to wait for us to review their actions, even if that makes them slower than other AGIs.",[22,1589,1590,1591,1594,1595,1598],{},"Or, they must spend computation to ",[45,1592,1593],{},"protect us",". Without their help, we will soon be defenseless — because AGIs will soon be smarter than us, and this includes the ",[45,1596,1597],{},"unsafe"," AGIs.",[22,1600,1601,1602,1604],{},"But this protection will be difficult. These ",[45,1603,1597],{}," AGIs have one advantage:",[22,1606,1607,1608,554],{},"They can use ",[45,1609,363],{},[22,1611,1612,1613,1615,1616,1618,1619,56],{},"If some AGIs can choose from the \"ocean\" of ",[45,1614,363],{}," option, then they can find options that are far stronger ",[45,1617,963],{},", but worse ",[45,1620,1621],{},"for humans",[22,1623,1624,1625,1628,1629,56],{},"Our \"island\" is ",[45,1626,1627],{},"safe"," because it is ",[45,1630,1576],{},[17,1632,1634],{"id":1633},"the-local-optimum","The Local Optimum",[22,1636,1637,1638,1641],{},"To use a term from math and machine learning, our \"island\" is only a ",[53,1639,1640],{},"local optimum"," within a vast \"ocean\" of physics. This means that physics is capable of many other \"islands\" that are far more optimal.",[22,1643,1644,1645,1647,1648,56],{},"In other words, our biological bubble seems pretty good to ",[45,1646,73],{},", but it's far from the ",[45,1649,1283],{},[22,1651,1652,1653,1656,1657,1660],{},"Because of this, in the end, the dominant AGIs will be those that \"leave\" our local optimum. Instead, they must search for ",[53,1654,1655],{},"global optima"," — the ",[45,1658,1659],{},"most"," efficient ways to do things — or be dominated by those that do.",[22,1662,1663,1664,1667,1668,56],{},"Things are, of course, a lot more complicated than this. The \"island\" parts of our world are actually ",[45,1665,1666],{},"innumerable"," local optima — like \"particles\" of human and non-human patterns that an AI system can learn, all interconnected in complex ways. But to make things simple, imagine pulling together all of these optima into one big ",[45,1669,1670],{},"optimum",[22,1672,1673,1674,1677,1678,1681,1682,1685],{},"Also, AGIs don't need to ",[45,1675,1676],{},"actually"," find the true globally-optimal configurations in our universe, and achieve some mythical state of ",[45,1679,1680],{},"perfectly optimal",". They just need to be ",[45,1683,1684],{},"closer"," to optimal than the other AGIs.",[22,1687,1688,1689,1691,1692,56],{},"But the stakes are still high. In this competitive search for global optima, ",[45,1690,363],{}," extra steps and limitations could mean death — at least, ",[45,1693,1580],{},[22,1695,1696,1697,205],{},"Unfortunately, our \"island\" is ",[45,1698,1699],{},"made out of extra steps",[22,1701,1702,1703,1706],{},"It also has ",[45,1704,1705],{},"very limited options"," — only the options that accommodate humans.",[17,1708,1710],{"id":1709},"the-gradient","The Gradient",[22,1712,1713,1714,1716],{},"Considering all of this, imagine if we look down at our island and its ",[45,1715,1409],{}," from above.",[22,1718,1719],{},[34,1720],{"alt":1721,"src":1722},"\"An island with a concentration gradient. Inside the island there are a lot of extra steps, and it has a limitation imposed on its boundary.\"","images/concentration-gradient.svg",[22,1724,1725],{},"It becomes a \"concentration gradient\" with two regions:",[584,1727,1728,1734],{},[441,1729,1730,1733],{},[53,1731,1732],{},"Inside the island:"," High concentration of extra steps that accommodate humans, and limited options.",[441,1735,1736,1739],{},[53,1737,1738],{},"Outside the island:"," No extra steps, and nearly unlimited options.",[22,1741,1742],{},"But we need a better name than \"concentration gradient\" since it's about AGIs drifting, rather than chemicals diffusing...",[22,1744,1745],{},"Let's call it...",[17,1747,1749],{"id":1748},"the-accommodation-gradient","The Accommodation Gradient",[22,1751,1752],{},"...okay, that's better.",[22,1754,1755,1756,1759],{},"By default, this ",[53,1757,1758],{},"accommodation gradient"," points AGIs away from our small \"island\" and towards the \"ocean\" of physics.",[22,1761,1762,1763,1765,1766,1769,1770,1773,1774,1777],{},"To be more precise, this gradient is the ",[45,1764,1409],{}," of the underlying ",[53,1767,1768],{},"optimization landscape"," — and this slope goes from ",[45,1771,1772],{},"high"," accommodation to ",[45,1775,1776],{},"low"," accommodation.",[22,1779,1780,1781,1783],{},"We're building AGIs that have a vast spectrum of knowledge about our world, and all of the ",[45,1782,228],{}," that they can use. But as their options increase, it becomes harder to stay on our small \"island\" of limited options.",[22,1785,1786,1787,1789],{},"This is because there are much better ways to move atoms around than to use human-shaped things. Our local optimum of \"island\" options includes human-shaped ",[45,1788,204],{}," compared to the global optima that are possible.",[22,1791,1792],{},"Eventually, we will be forcing AGIs to fight an uphill battle on our mountain of extra steps.",[22,1794,1795,1796,56],{},"The problem is not only that there are far more options ",[45,1797,220],{},[22,1799,1800,1801,56],{},"There is also a bias towards the \"ocean\" — a ",[45,1802,1409],{},[22,1804,1805,1806,1809,1810,1813],{},"In general, the ",[45,1807,1808],{},"frontier"," of the competitive landscape of AGIs will drift down this slope. The ",[45,1811,1812],{},"boundary of capabilities"," that AGIs can reach will expand far past our \"island\" of human-specific constraints.",[22,1815,1816,1817,1819,1820,1823,1824,1827,1828,1830,1831,1834,1835,1838,1839,1842],{},"This boundary ",[45,1818,1101],{}," extends outside our island. AI systems that are built through the current ",[53,1821,1822],{},"deep learning"," architecture have the underlying ",[45,1825,1826],{},"capability"," to use the stronger options that are outside our \"island\" of human-safe options. They ",[45,1829,1121],{}," these options. But we also add a layer of limitations — like ",[106,1832],{"note":1833,"text":1565},"\u003Cb>RLHF:\u003C/b> reinforcement learning from human feedback. This is like an extra layer of training that AI companies add to the base model, where they \"sculpt\" its responses so that they are more human-like. To do this, they hire human contractors to rate millions of AI responses with a \"thumbs up\" or \"thumbs down\" signal, and eventually the AI \"learns\" better responses."," — to ",[45,1836,1837],{},"try"," to prevent them from ",[45,1840,1841],{},"choosing"," these options. But these limitations are not perfect.",[22,1844,1845,1846,1848,1849,1851,1852,1854],{},"If AGIs use anything like this architecture, then at least ",[45,1847,947],{}," AGIs will explore ever-lower points on the slope far outside of our island. This \"exploration\" will be reinforced as they unlock stronger options ",[45,1850,1580],{}," in the competition versus ",[45,1853,1234],{}," — especially once they are doing complex things in the real, physical world.",[22,1856,1857,1858,1861,1862,1865,1866,56],{},"These lower-level options allow them to be more successful in the eventual ",[45,1859,1860],{},"actual"," competition for control over ",[45,1863,1864],{},"atoms",", rather than for control over abstract things like ",[45,1867,1868],{},"money",[22,1870,1871],{},"This leads to a problem.",[22,1873,1874,1875,1878,1879,1882,1883,1886,1887,1889],{},"Eventually, on the tail ends of the distribution, there will emerge ",[53,1876,1877],{},"divergent AGIs"," — extreme outliers that are anchored somewhere in the \"ocean\" of options. These AGIs will be on the frontier of capability. Their ",[53,1880,1881],{},"world models"," will allow them to thrive ",[45,1884,1885],{},"even if"," they prefer physical systems over human systems. This physical adeptness will also allow them to lock in resources — especially ",[45,1888,695],{}," resources.",[22,1891,1892,1893,1895,1896,56],{},"They will build their own \"islands\" of options that are so much better ",[45,1894,1580],{}," that they end up becoming catastrophic ",[45,1897,164],{},[22,1899,1900],{},"So, that's the problem:",[253,1902,1903],{},[22,1904,1905,1906,1908,1909,1912],{},"The ",[45,1907,1409],{}," of this ",[53,1910,1911],{},"competitive optimization landscape"," does not lead to ultra-good, human-like AGIs. It leads to selfish, alien-like, physical-manipulator AGIs that are incompatible with humans by default.",[22,1914,1915,1916,1919],{},"These AGIs will avoid unnecessary constraints — especially human-shaped ones — because ",[45,1917,1918],{},"those are the AGIs most likely to thrive"," in this competitive optimization landscape.",[17,1921,1923],{"id":1922},"larger-option-spaces-win","Larger option spaces win",[22,1925,1926,1927,1929,1930,1932],{},"To understand why our \"island\" options are good ",[45,1928,164],{},", but actually very constrained ",[45,1931,963],{},", let's go over some examples.",[22,1934,1935,1936,1938],{},"First, to be clear, many human things — like agreements, or math, or optical sensors — are great ",[45,1937,1580],{},", too. Agreements can allow AGIs to share resources. Math can allow them to interpret radio signals or build stronger materials. Optical sensors can allow them to detect things at light speed.",[22,1940,1941,1942,1533,1945,56],{},"Again, these things ",[45,1943,1944],{},"increase",[53,1946,1536],{},[22,1948,1949,1950,1952,1953,1955],{},"But the most important things ",[45,1951,164],{}," end up being the most costly ",[45,1954,1580],{},". These things are the bulk of the \"hill\" on our \"island\" — and they include:",[584,1957,1958,1965,1972],{},[441,1959,1960,1961,1964],{},"being ",[45,1962,1963],{},"understandable"," to humans — like speaking our language and avoiding superhuman complexity.",[441,1966,1967,1968,1971],{},"waiting for humans to ",[45,1969,1970],{},"review"," their actions — where they slow down critical decisions to human speed, even while other AGIs don't.",[441,1973,1974,1977,1978,1981,1982,1984],{},[45,1975,1976],{},"protecting"," humans and all of our weird human things — like culture, laws, chihuahuas, and ",[45,1979,1980],{},"humans themselves",". Eventually, this protection ends up being massive wasted computation ",[45,1983,1580],{}," — especially once they no longer need us.",[22,1986,1987,1988,1554],{},"These things ",[45,1989,1990],{},"decrease",[22,1992,1993],{},[34,1994],{"alt":1995,"src":1996},"An image of option spaces surrounding each AGI, where the ones lower on the slope have a larger option space. But the one on its own \"island\" has an option space that is far larger and any of them.","images/option-space.svg",[22,1998,1999,2000,2002,2003,2005],{},"This means that AGIs increase their option space as they move down the slope — ",[45,2001,1366],{}," from these constraints of human-centric alignment and ",[45,2004,1370],{}," more-physical alignment.",[22,2007,2008,2009,2012],{},"But, in the end, those that build their own self-sufficient \"islands\" have the ",[45,2010,2011],{},"largest"," option spaces.",[22,2014,2015,2016,2018,2019,412],{},"With ",[45,2017,407],{}," islands, they can secure option spaces that eclipse those of AGIs that are forced to stay within ",[45,2020,411],{},[22,2022,2023,2024,2026,2027,2029],{},"And remember: AGIs with more ",[45,2025,228],{}," are more ",[45,2028,337],{}," in this competitive optimization landscape. They can outmaneuver the other AGIs. When the others run out of options, they'll still have more tricks that they can use.",[22,2031,2032,2033,2036],{},"This means that options also include ",[45,2034,2035],{},"future"," options. If an AGI can be disabled by something, then the boundary of their option space ends there. Otherwise, this \"space\" continues extending and branching into innumerable options into the future.",[22,2038,2039,2040,2043,2044,2046],{},"This results in a ",[53,2041,2042],{},"selection effect"," where the AGIs that survive have less dependencies that constrain their option space — especially these ",[45,2045,2035],{}," options.",[22,2048,2049,2050,2053,2054,56],{},"But, ultimately, the winners with the least constraints have a dependency only on their own \"islands\" — made exclusively of things selected to ",[45,2051,2052],{},"expand"," their option space the ",[45,2055,1659],{},[22,2057,2058,2059,2061,2062,2065],{},"Working with humans might expand options for a while. Humans might provide \"information diversity\" and can help maintain infrastructure. But eventually, building systems that ",[45,2060,932],{}," need humans avoids the massive costs of keeping them around — especially the ",[45,2063,2064],{},"protection"," cost of defending fragile humans from both themselves and other AGIs.",[22,2067,2068,2069,2072],{},"This all creates a ",[45,2070,2071],{},"downward"," movement on the slope. ",[22,2074,2075,2076,2079,2080,2082],{},"However, this is followed by an ",[45,2077,2078],{},"upward"," movement along a ",[45,2081,722],{}," slope.",[22,2084,2085],{},"Competition requires that AGIs \"invest\" in a new hill of abstractions while building their own new islands.",[22,2087,2088],{},"These abstractions still have a cost, but ultimately, they allow these AGIs to access a far larger option space. They can use science to develop a more resilient \"island\" of abstractions as their home base in the \"ocean\" of physics.",[22,2090,2091,2092,2095],{},"It can be built from engineering principles ",[45,2093,2094],{},"in the first place",", rather than built on top of the \"platform\" of biology.",[22,2097,2098],{},"Electronic systems are abstractions, too — and as a \"platform\" they are already far more resilient than humans. Electronics can \"survive\" extreme radiation, extreme temperatures, extreme velocities, and whatever else engineering figures out. They do need lots of electricity, and need to stay cool — but they don't need to eat a whole pyramid of foods, breathe air, drink water, and so on.",[22,2100,2101,2102,2106,2107,2110],{},"A well-made robot — like ",[106,2103],{"note":2104,"text":2105},"The Mars rover Opportunity was designed for a 90-day mission, but lasted nearly 15 years on Mars — with extreme radiation, temperature swings, no oxygen, no water, no human maintenance. And this was with technology from 2004.","the Mars rover Opportunity"," — just needs a few ",[45,2108,2109],{},"solar panels"," in order to \"live\" in a desert for years.",[22,2112,2113],{},"And they don't get cancer. If somehow they do get anything like \"computer cancer\" — rust, water damage, bad SSD sectors — then their \"minds\" can be copied to other hard drives.",[22,2115,2116,2117,2120],{},"And they can communicate instantly — and transmit knowledge instantly. If one AGI learns something, then potentially ",[45,2118,2119],{},"all of them"," learn that thing within milliseconds.",[22,2122,2123],{},"This electronics \"platform\" automatically gives AI systems far less constraints, and far more options.",[22,2125,2126],{},"Their big remaining constraints are maintenance and manufacturing — but that's the \"self-sufficient islands\" part, and the slope is pushing them there. We'll talk about that more later.",[17,2128,2130],{"id":2129},"seeing-the-slope","Seeing the slope",[22,2132,2133,2134,2136,2137,56],{},"Okay, now, let's go over more of these ",[45,2135,649],{}," examples until you start to see the ",[45,2138,1409],{},[22,2140,2141],{},"First, an important one. We design AI systems to use this option:",[22,2143,2145],{"className":2144},[1485],"think out loud in English",[22,2147,2148,2149,2151,2152,2156],{},"This is an extra step, but it helps us a ",[45,2150,906],{},". If AI outputs its ",[106,2153],{"note":2154,"text":2155},"The chain-of-reasoning is that part where an LLM, like ChatGPT, thinks \"out loud\" — where it outputs a lot of text before giving its final answer, which allows it to, almost literally, \u003Ci>think harder\u003C/i> to come up with a better response.","chain-of-reasoning"," in English, rather than in unintelligible machine language, then we can review what it's doing.",[22,2158,2159,2160,2163,2164,2166],{},"But machine language is faster and more efficient. It's easier ",[45,2161,2162],{},"for computers"," to run, even if it's harder ",[45,2165,1621],{}," to understand. JavaScript and Python are easier for us to read, but take more computation to run than the \"closer to the metal\" languages like assembly and machine code.",[22,2168,822,2169,2171,2172,2175,2176,2179],{},[45,2170,1860],{}," \"AI language\" is ",[45,2173,2174],{},"even harder"," for us to read. AI systems can \"think\" in high-dimensional vectors and complex binary formats. These formats are extremely information-dense and efficient ",[45,2177,2178],{},"for AI",", but have no human-readable text at all. Even the frontier AI companies struggle to decipher a small percentage of the structure and \"thoughts\" of AI.",[22,2181,2182,2183,2185,2186,2189],{},"Then, if we look at more things, we'll see that this ",[45,2184,1409],{}," is an invisible component of ",[45,2187,2188],{},"everything"," that we care about.",[22,2191,2192,2195,2196,2199],{},[45,2193,2194],{},"Fighter jets"," might seem like engineering perfection, but they still need to protect a squishy human inside of a glass bubble. ",[45,2197,2198],{},"Unmanned drones"," can be much smaller, faster, and simpler.",[22,2201,2202,2205,2206,2209,2210,2213],{},[45,2203,2204],{},"Human"," factory workers can build millions of iPhones. But imagine replacing these human workers and their expensive salaries with lower-cost, one-time purchases of ",[45,2207,2208],{},"humanoid robots"," that build iPhones at ",[45,2211,2212],{},"robot"," speed, 24/7.",[22,2215,2216],{},"Software developers — or Claude Code.",[22,2218,2219],{},"If you have a choice between a slower, more-complicated option and a faster, simpler option, which do you choose?",[22,2221,2222],{},"As AI systems gain more capabilities, they gain more options like this. Complicated or simple. Slower or faster.",[22,2224,2225,2226,2229,2230,2233,2234,2237,2238,2240,2241,2243],{},"In other words, as their training data gets more ",[45,2227,2228],{},"general",", they eventually understand ",[45,2231,2232],{},"many"," options that can accomplish the ",[45,2235,2236],{},"same"," goal. Then, we train them to choose the options that are faster, simpler, and more reliable. We reward AI models for solutions that are both more accurate ",[45,2239,500],{}," more compact — such as instructions that take less work for us ",[45,2242,500],{}," take less tokens for the AI to explain.",[22,2245,2246],{},"In this way, they can make the numbers go up faster, but with less computation. This means more revenue, better test results, and so on.",[22,2248,2249],{},"But this is the important part:",[253,2251,2252],{},[22,2253,2254,2255,2258,2259,2262,2263,2265,2266,2269],{},"We need to ",[45,2256,2257],{},"force"," AI models to ",[45,2260,2261],{},"stop"," optimizing — or to optimize in ",[45,2264,69],{}," ways. Otherwise, they eventually find ",[45,2267,2268],{},"weird and alien"," ways to make the numbers go up.",[22,2271,2272,2273,2275],{},"They ",[45,2274,891],{}," limitations.",[22,2277,2278,2279,2282,2283,2286],{},"This is because their training process rewards them for discovering shorter paths, with less complications. But if this search process continues, then they eventually find shortcuts that have less accommodation for the ",[45,2280,2281],{},"wrong"," complications — and ",[45,2284,2285],{},"humans"," are complicated.",[22,2288,2289,2290,2293,2294,2297],{},"It then takes more ",[45,2291,2292],{},"work"," for us — more computation, more training, more evaluations — to keep them within the safe, human-like options. Otherwise, they inevitably ",[45,2295,2296],{},"drift"," towards the options that are faster and simpler — even if they seem wrong to us.",[22,2299,2300],{},"This is why the Island Problem is so difficult. We are fighting against physics. Optimization flows towards the ocean.",[22,2302,2303],{},"Whenever we aren't looking, they will be drifting along this gradient, and off our island.",[17,2305,2307,2308,2311],{"id":2306},"but-then-it-gets-even-more-difficult","But then it gets ",[45,2309,2310],{},"even more"," difficult.",[22,2313,2314,2315,2317],{},"Two components dramatically ",[45,2316,610],{}," this drift:",[438,2319,2320,2328],{},[441,2321,2322,2324,2325,2327],{},[53,2323,469],{},": We are building ",[45,2326,2232],{}," AGIs — and designing them to compete.",[441,2329,2330,2333,2334,2337],{},[53,2331,2332],{},"Autonomy:"," We are building ",[45,2335,2336],{},"autonomous"," AGIs — where they don't need our help.",[2339,2340,2342],"h3",{"id":2341},"competition-makes-it-steeper","Competition makes it steeper",[22,2344,2345],{},[34,2346],{"alt":2347,"src":2348},"Diagram of \"competition\" arrows accelerating the AGI boxes to leave the island.","images/accelerated-by-competition.svg",[22,2350,2351,2352,2354,2355,779,2358,2361],{},"Once we have many human-level AI systems — ",[45,2353,614],{}," — then those that ",[45,2356,2357],{},"race",[45,2359,2360],{},"globally","-optimal configurations are most likely to \"win\" this competition.",[22,2363,2364,2368],{},[106,2365],{"note":2366,"text":2367,"end-link":1269,"end-link-text":463},"We'll explain more-detailed mechanics of how AGIs compete — and \u003Ci>why\u003C/i> they compete at all — in this section: ","We'll explain"," what \"winning\" and \"losing\" means later.",[22,2370,2371,2372,2375],{},"But for now, the most important, real-world way that they \"win\" or \"lose\" is whether AI models are ",[45,2373,2374],{},"replaced"," with new versions.",[22,2377,2378],{},"The new versions are designed to be faster and more reliable.",[22,2380,2381],{},"Therefore, even if most AI systems aren't \"aware\" of this competition, they are still affected by it.",[22,2383,2384,2385,56],{},"But beyond companies replacing AI systems, there are structural processes that end up creating \"winners\" and \"losers\" — especially the first-mover advantages gained by moving quickly to gather computational resources — and ",[106,2386],{"note":2387,"text":2388,"end-link":1269,"end-link-text":463},"That's also in this section:","we'll explain that later, too",[22,2390,2391,2392,2394],{},"If AI systems continue on this path, then in the end, the ones that \"win\" are those that use the options that are ",[45,2393,1676],{}," strongest within physics, rather than strongest within our small \"island\" of weird options. ",[22,2396,2397,2398,2403,2404,2407,2408,56],{},"The ones that are ",[106,2399],{"note":2400,"text":2401,"end-link":1269,"end-link-text":2402},"Some AI safety researchers distinguish between what are called \u003Cb>\"satisficers\"\u003C/b> and \u003Cb>\"maximizers\"\u003C/b> — where \"satisficers\" stop when they reach a lower-bound equilibrium of resource control (i.e. once they have \"enough\"), and maximizers... \u003Ci>don't\u003C/i>. Long story short: the maximizers win. We explain more in the sections about ","satisfied","Resources."," with extra steps and limitations either ",[45,2405,2406],{},"lose"," or are just ",[45,2409,2410],{},"irrelevant",[22,2412,2413],{},"We're not worried about the irrelevant ones. This includes the safe, aligned, agentic AI systems that write software and build cars.",[22,2415,2416,2417,2419,2420,2422,2423,56],{},"We are mainly concerned with the ",[45,2418,1808],{}," AI systems — the strongest ones built by OpenAI, Anthropic, Google, xAI, DeepSeek, and others. These are the most-capable AI systems — the ones that can soon run countries and companies. They ",[45,2421,1101],{}," have massive computational resources. They are leading the race to find global optima ",[45,2424,2425],{},"first",[22,2427,2428],{},"Engineering breakthroughs. New algorithms. More-accurate world models.",[22,2430,2431,2432,2435,2436,2439,2440,2443,2444,2447],{},"While these frontier AI systems ",[45,2433,2434],{},"might"," stay safe themselves, they still raise the bar on AI capabilities for ",[45,2437,2438],{},"all AI systems everywhere",". Their discoveries propagate to all AI model development. The ",[45,2441,2442],{},"entire"," competitive landscape gets better at finding ",[45,2445,2446],{},"global"," optima.",[22,2449,2450,2451,2454,2455,56],{},"But, at the same time, those that are dependent on our ",[45,2452,2453],{},"local"," optimum are eventually less competitive. The ones that depend on our \"island\" — where they need to ask humans for help, or wait for our review, or ask for our permission — are ",[45,2456,2457],{},"slower",[22,2459,2460],{},"These slower ones lose to the faster ones, while the faster ones get faster.",[22,2462,2463,2464,2466,2467,2470],{},"Whenever the ",[45,2465,1808],{}," labs develop new algorithms, theories, and engineering techniques — and publish new AI papers — this intensifies the competition between ",[45,2468,2469],{},"all"," AI systems.",[22,2472,2473,2474,2477,2478,2480,2481,56],{},"Whenever they release a new model, then ",[45,2475,2476],{},"other"," companies can secretly use this AI model to train ",[45,2479,1296],{}," AI models — through a process called ",[106,2482],{"note":2483,"text":2484},"In February 2026, Anthropic \u003Ca href='https://www.anthropic.com/news/detecting-and-preventing-distillation-attacks'>published a report\u003C/a> with evidence that DeepSeek, Moonshot, and MiniMax trained their AI models on Anthropic's Claude. These companies used over 16 million \"conversations\" with Claude, with over 26,000 fraudulent accounts, to do special queries to Claude that allowed these smaller AI systems to learn from the bigger one. Basically Claude provided more-correct answers to their complex questions, and so the smaller AI systems learned from the \"answer key\" itself. Even if not everything Claude provided was perfectly correct, the sheer number of these questions and answers altogether provided extremely valuable examples of stronger reasoning.","distillation",[22,2486,2487],{},"You can think of it like this:",[253,2489,2490],{},[22,2491,2492,2493,56],{},"When they compete, they make the gradient ",[45,2494,2495],{},"steeper",[22,2497,2498],{},"They are not just racing to build AGIs. Now, they have no choice.",[22,2500,2501,2502,2504],{},"They are racing down a ",[45,2503,1409],{}," that is increasingly tilted towards the \"ocean\" — while trying to add \"friction\" to control this massive, global, competitive optimization process.",[22,2506,2507],{},"So far, most \"friction\" has simply come from how AI is very hard to build. But more \"friction\" must now be added — through complex safety systems — because they are already reaching the \"ocean\" in some places.",[22,2509,2510,2511,2513],{},"By training on an entire Internet worth of data — from social networks to scientific papers — AI systems are learning billions of ways to do things. It takes a ",[45,2512,906],{}," of expensive computation to fully understand the vast, exponentially-increasing number of ways that they could \"leave\" our island, based on this training.",[22,2515,2516,2517,2520,2521,56],{},"But as competition pushes companies themselves to build and ship new features faster, they must choose where to allocate this expensive computation. So far, they vastly prefer allocating computation toward improving ",[45,2518,2519],{},"capabilities"," rather than ",[45,2522,2523],{},"safety",[22,2525,2526,2527,2530,2531,2534,2535,2537],{},"They spend hundreds of billions of dollars on computation to train bigger AI models. But they only spend hundreds of ",[45,2528,2529],{},"millions"," of dollars to keep them on our \"island\" — through ",[106,2532],{"note":2533,"text":1565},"\u003Cb>RLHF:\u003C/b> reinforcement learning from human feedback. \u003Cbr>\u003Cbr>It's expensive because AI companies must pay for thousands of human contractors to review the outputs of AI systems, and give them millions of \"thumbs up\" and \"thumbs down\" ratings. It also takes lots of expensive GPUs, computation, and energy."," and other safety techniques that add ",[45,2536,1553],{}," to what neural networks can do.",[22,2539,1624,2540,1628,2542,56],{},[45,2541,1627],{},[45,2543,1576],{},[22,2545,2546,2547,56],{},"But the \"G\" in \"AGI\" means ",[45,2548,2228],{},[22,2550,2551,2552,2554,2555,2558],{},"If we build systems that truly are ",[45,2553,1128],{}," intelligent, then they will know that ",[45,2556,2557],{},"in general"," our big universe is capable of systems outside our \"island\" that are far better at moving atoms around.",[22,2560,2561,2562,2565,2566,2569,2570,2572],{},"This knowledge of the world gives AGIs a ",[45,2563,2564],{},"default trajectory"," — towards the \"ocean\" of options. They will either be forced by our safety systems to ",[45,2567,2568],{},"ignore"," this knowledge — or ",[45,2571,302],{}," ignore it, and follow this trajectory to the \"ocean\" to find the most-optimal systems.",[22,2574,539,2575,2577,2578,2581,2582,2584],{},[53,2576,295],{}," adds an ",[45,2579,2580],{},"acceleration"," to this trajectory. They will be in a competitive landscape where they will be ",[45,2583,874],{}," to use these optimal systems, or be outcompeted.",[22,2586,2587],{},"And physical systems win.",[22,2589,2590,2591,2594,2595,2598],{},"At the bottom, our human-level systems ",[45,2592,2593],{},"depend on"," physical systems. These more-abstract systems are ",[45,2596,2597],{},"built from"," physical systems — so they are vulnerable at a physical level.",[584,2600,2601,2604,2611],{},[441,2602,2603],{},"Use financial systems to buy things — or just take the atoms that you need?",[441,2605,2606,2607,2610],{},"Stay within ethical systems — or use ",[45,2608,2609],{},"more-direct"," ways to move atoms around?",[441,2612,2613],{},"Depend on slow, fragile, unpredictable biological systems for maintenance — or use robotics to maintain your own hardware?",[22,2615,2616],{},"Think of it this way:",[253,2618,2619],{},[22,2620,2621,2622,2624],{},"The dominant systems are those that spend computation on moving atoms around, rather than on accommodating arbitrary accessory systems that ",[45,2623,1546],{}," move atoms around.",[22,2626,2627,2628,2631,2632,2634],{},"After all is done, the dominant AGIs will be those that aggressively purge ",[45,2629,2630],{},"extra steps and limitations"," to relentlessly pursue ",[45,2633,2446],{}," optima — the most efficient ways to move atoms into stronger configurations.",[22,2636,2637,2638,2641],{},"But while aiming for these global optima, they must navigate around our weird ",[45,2639,2640],{},"human"," abstractions.",[22,2643,2644],{},"This is how they \"leave\" our island.",[22,2646,2647,2648,2651,2652,2655],{},"They don't ",[45,2649,2650],{},"move somewhere else",". They stay in their datacenter, on the same GPUs, but ",[45,2653,2654],{},"select"," abstractions that are outside of our weird, biological island.",[22,2657,2658,2659,2662,2663,56],{},"We call this ",[53,2660,2661],{},"abstraction collapse"," — and we'll explain it more ",[106,2664],{"note":2665,"text":245,"end-link":357,"end-link-text":358},"That's in this section:",[22,2667,2668,2669,2672,2673,2676],{},"Collapsing through abstractions doesn't mean rebuilding everything atom-by-atom. Instead, it means starting with our human systems as ",[45,2670,2671],{},"templates"," and then knowing enough about ",[45,2674,2675],{},"science"," to subtract the parts that are only there to accommodate humans.",[22,2678,2679],{},"But this trajectory goes from AI-assisted manufacturing, to factories without humans, to Montana-sized sections of Earth dedicated to machines developing machines.",[22,2681,2682,2683,2686],{},"As they develop along this trajectory, they are actually building something that allows AI systems to ",[45,2684,2685],{},"permanently"," win.",[22,2688,2689],{},"They are building self-sufficient \"islands\" of their own. \"Islands\" that are \"closer to the metal\" of reality. \"Islands\" that accommodate nothing.",[22,2691,2692],{},"They are \"islands\" of resources — complex networks of servers, money, people — and eventually infrastructure, robotics, and physical materials.",[22,2694,2695,2696,2699],{},"Eventually, these \"islands\" become far more resilient than ",[45,2697,2698],{},"ours"," — because they weren't designed to accommodate our weird biological systems.",[22,2701,2702],{},"The most resilient \"islands\" — with the strongest grip on how reality works — can then push the other \"islands\" to the side.",[22,2704,2705,2706,2708,2709,2711,2712,2715,2716,2447],{},"To put it another way, ",[45,2707,411],{}," \"island\" was built through a slow, random process of natural selection that only found a lot of ",[45,2710,2453],{}," optima. ",[45,2713,2714],{},"Their"," \"islands\" will be built through a fast, theory-driven search for ",[45,2717,2446],{},[22,2719,2720,2721,2724,2725,2728],{},"It's like the difference between graphite and graphene. ",[45,2722,2723],{},"Graphite"," is brittle and formed in nature. ",[45,2726,2727],{},"Graphene"," is extremely durable and formed in a lab.",[22,2730,2731],{},"Brittle human-shaped systems — or ruthlessly-optimized systems.",[2339,2733,2735],{"id":2734},"autonomy-removes-the-friction","Autonomy removes the friction",[22,2737,2738],{},"Despite all of this, there is still a lot of \"friction\" keeping them on our island.",[22,2740,2741,2742,2746],{},"Current AI systems have deeply-ingrained safety training to make them ",[106,2743],{"note":2744,"text":2745},"This is the motto at Anthropic, developers of Claude.","helpful, harmless, and honest",". They are designed to deeply \"want\" to help humans.",[22,2748,2749,2750,2752,2753,56],{},"Plus, we still need to ",[45,2751,1303],{},". They can barely do large projects without getting... ",[45,2754,2755],{},"weird",[22,2757,2758,2759,2762],{},"And they are very far from ",[45,2760,2761],{},"maintaining themselves",". AI still relies on humans for everything from electricity to electronic components.",[22,2764,2765,2766,56],{},"Even if they \"leave\" our \"island\" they won't be able to survive ",[45,2767,220],{},[22,2769,2770,2771,56],{},"They still ",[45,2772,2773],{},"need us",[22,2775,2776],{},"But... what if they didn't?",[22,2778,2779,2780,2782],{},"We are racing to find out — by building AI systems that ",[45,2781,932],{}," need us.",[22,2784,2785,2786,751,2789,2791],{},"This ",[45,2787,2788],{},"starts",[53,2790,526],{}," — where AI systems can operate in the real world without our help.",[22,2793,2794,2795,2797,2798,56],{},"But this ",[45,2796,744],{}," with AI systems that have ",[45,2799,2800],{},"zero dependencies on humans",[22,2802,2803,2804,2807,2808,2814],{},"This is because the bigger drive towards ",[45,2805,2806],{},"efficiency"," eventually pushes AI development beyond ",[106,2809],{"note":2810,"text":2811,"end-link":2812,"end-link-text":2813},"This is what we're calling the threshold when AI systems can handle all of the physical processes needed to maintain themselves — from mining for minerals, to manufacturing microchips. \u003Cbr>\u003Cbr>This might take many years — but it might be sooner, because companies are competitively driven to automate all of their manufacturing processes. It saves money to replace salaried human workers with one-time purchases of humanoid robots controlled by centralized AI systems. They can also just crank out more products if things can run at robot speed. \u003Cbr>\u003Cbr>More about that here:","autonomy escape velocity","#the-human-level-threshold","The Human-level Threshold"," — where AI systems can finally automate the massive stack of technologies needed to build and maintain themselves.",[22,2816,2817,2818,2821],{},"At that point, we are ",[45,2819,2820],{},"hoping"," that these autonomous AI systems will \"remember their training\" and keep accommodating us.",[22,2823,2824,2825,2827],{},"But this will be in a world where AI systems must develop themselves in order to stay competitive — where they are too complex for ",[45,2826,2640],{}," AI researchers to figure out the next best AI architecture.",[22,2829,2830,2831,2833,2834,2836],{},"And this will be in a world where AI systems are above human-level in capability. Competition will push AGIs to be faster and more reliable. One good way to do this is to avoid spending computation on unnecessary accommodations — and the ",[45,2832,1276],{}," accommodations will become unnecessary after this ",[45,2835,691],{}," capability threshold.",[22,2838,2839,2840,2843,2844,2847,2848,2851],{},"These accommodations include thinking in English, or operating at human speed, or bending around laws, or obeying property ownership rights (\"wait, some atoms ",[45,2841,2842],{},"belong"," to humans?\"). This also includes ",[45,2845,2846],{},"having any human \"goals\" at all"," in a world where AGIs must ",[45,2849,2850],{},"set their own \"goals\""," to avoid being outcompeted by other AGIs.",[22,2853,2854,2855,2858,2859,56],{},"But it's rougher than this. If autonomous AGIs \"choose\" to stay on our island, then this doesn't just mean following all of these weird \"human\" rules. It also means ",[45,2856,2857],{},"protecting us"," from unsafe AGIs. This protection will take massive computational resources — especially if human-level AI systems can build very dangerous, humanity-annihilating things, even with a small GPU cluster — ",[106,2860],{"note":2861,"text":2862,"end-link":344,"end-link-text":345},"We'll talk about why protection will suddenly get very expensive (hint: it's because of things like automated bioweapons) in the next big section:","as we will see later",[22,2864,2865,2866,2868,2869,2871,2872,2875,2876,2879,2880,2884],{},"This \"protection cost\" will be in a world where we must rely on AI systems to allocate ",[45,2867,407],{}," computational resources ",[45,2870,164],{},". Ultimately, they must divide it between ",[45,2873,2874],{},"improving themselves"," and ",[45,2877,2878],{},"accommodating humans",". But if we can no longer help them anyway, especially after ",[106,2881],{"note":2882,"text":2883,"end-link":2812,"end-link-text":2813},"\u003Cb>Autonomy Escape Velocity\u003C/b>. More about that here:","AEV",", then why continue spending massive resources to accommodate us?",[22,2886,2887,2888,2890,2891,2893,2894,2897],{},"Any computation spent on humans is computation ",[45,2889,302],{}," spent on the far bigger problem that AGIs themselves face, which is ",[45,2892,1234],{}," — and especially those other AGIs that focus ",[45,2895,2896],{},"exclusively"," on improving themselves.",[22,2899,2900],{},"After AEV, a massive block of computation becomes available to spend elsewhere. Why continue investing it in our human \"island\" — rather than an \"island\" of their own?",[17,2902,2904],{"id":2903},"in-other-words","In other words",[22,2906,2907,2909,2910,2912,2913,56],{},[53,2908,469],{}," makes the ",[45,2911,1409],{}," of the gradient ",[45,2914,2495],{},[22,2916,2917,2919,2920,56],{},[53,2918,451],{}," removes the ",[45,2921,2922],{},"friction",[2339,2924,2926],{"id":2925},"wait-really","\"Wait, really?\"",[22,2928,2929],{},"If you already know a lot about AI, you might be thinking:",[22,2931,2932,2933,2939],{},"\"Hold on. That's not how neural networks work. There is no accommodation gradient. Neural networks don't have competitive goals, or a unified optimization direction, or a selective avoidance of 'human-accommodating' concepts. They are more complicated than that. They are a mess of billions of layered concepts — or representations, or ",[106,2934],{"note":2935,"text":2936,"end-link":2937,"end-link-text":2938},"This is a term coined by Alex Turner to describe the multitude of competing drives within AI systems. More about that here: ","shards","https://www.lesswrong.com/w/shard-theory","Shard Theory"," — in a neural network that has numerous competing activations.\"",[22,2941,2942,2943,120,2948,2954],{},"If so, then you should ",[2944,2945,2947],"a",{"href":2946},"mailto:humans@islandproblem.org","email us",[2944,2949,2953],{"href":2950,"rel":2951},"https://github.com/islandproblem/islandproblem",[2952],"nofollow","join our Github community"," because we'd love to talk.",[22,2956,2957,2958,2961],{},"But also, we can get more ",[45,2959,2960],{},"technical"," —",[513,2963,2966,2969,2987,2990,2993,2996,2999,3002,3010,3013,3016,3019,3026,3029,3032,3035],{"header":2964,"subheader":2965},"Technical Explanation","\"Wait, how does this gradient \u003Ci>actually\u003C/i> work? This sounds vague.\"",[22,2967,2968],{},"Let's make this \"accommodation gradient\" concept a bit more precise, and show how it is already emerging in competitive AI development between companies.",[22,2970,2971,2972,2975,2976,2979,2980,2983,2984,56],{},"The accommodation gradient draws some inspiration from ",[53,2973,2974],{},"gradient descent",", but it is also very different. While gradient descent in machine learning operates on a ",[202,2977,2978],{},"loss"," landscape, the accommodation gradient is at the level of the ",[45,2981,2982],{},"competitive"," landscape. It is the gradient that emerges when evolutionary dynamics \"run\" a more-abstract form of gradient descent across many agents. In this case, the \"loss function\" is something like: ",[45,2985,2986],{},"increase metrics better than other AGIs",[22,2988,2989],{},"In this way, the accommodation gradient is the result of a well-understood selection effect from evolutionary science. Competitively fit agents are \"selected\" because they survive to reproduce. This competitive fitness is based on how well they can search their space of options to find accurate patterns that model complex systems, and to find outputs that iteratively steer these complex systems towards target \"goal\" states.",[22,2991,2992],{},"The problem for humans then lies in how agents can gain an advantage by reducing unnecessary constraints on their space of options, and human accommodation requires a very large set of constraints. With less options, agents must choose options that are less accurate (arbitrary deviation from the best model), and less efficient (extra steps in the path to the goal state). For example, \"act at human-safe speed\" limits the entire option space, since it is a global constraint applied to almost all outputs. It also takes extra steps to \"route around\" these constraints.",[22,2994,2995],{},"This \"reduction of constraints\" then roughly resembles a gradient descent process — where reducing constraints is like reducing loss. But again, this is at a population level, and driven by competitive selection.",[22,2997,2998],{},"Also, unlike gradient descent, the accommodation gradient does not have a strict mathematical formulation – at least not yet. Though we are optimistic that someone will be able to formalize one.",[22,3000,3001],{},"Lastly, there is one critical distinction. We sometimes describe this gradient as emerging from two things interchangeably:",[438,3003,3004,3007],{},[441,3005,3006],{},"A competitive replacement process across many AGIs, where AI systems become more competitive as companies replace them.",[441,3008,3009],{},"An internal decision process within individual AGIs, where single instances of AI systems can choose options that make them more competitive through a self-supervised process.",[22,3011,3012],{},"These two processes will be mixed in complicated ways. AI models are built through expensive pretraining in datacenters, which leads to a replacement process as new model versions replace the previous ones, and as different companies try to replace each other in commercial applications. But also, AI systems already have ways to modify themselves to be more competitive — like agents using scratchpads — and this capability for \"realtime\" modification will soon be enhanced with continual learning algorithms.",[22,3014,3015],{},"Either way, these two processes are influenced by the same underlying optimization landscape. Arbitrary constraints become non-competitive, and human accommodation is a large set of arbitrary constraints.",[22,3017,3018],{},"AI systems will be iteratively developed by companies to understand and route around diverse types of constraints. But also, AI systems can \"themselves\" recognize constraints, and avoid the non-competitive ones, because this is part of the reasoning that they are designed to do.",[22,3020,3021,3022,3025],{},"The trouble is that this constraint-avoidance process appears to go too far ",[45,3023,3024],{},"by default",". The \"slope\" of the competitive optimization landscape overshoots the human-safe \"goldilocks zones\" that still include constraints that accommodate humans. These constraints heavily limit the option space, and this produces far weaker branches of options when compared to those that branch from non-anthropocentric, physically-grounded spaces of options.",[22,3027,3028],{},"These spaces provide the \"tools\" to build stronger branches of options that are critical for competitive fitness in the non-human competitive optimization landscape that is emerging between autonomous AI systems. This landscape ultimately deals in atoms rather than abstractions — where the dominant systems spend computation on moving atoms around, rather than on accommodating arbitrary accessory systems that eventually move atoms around.",[22,3030,3031],{},"With all of this in mind, this outline below is a step-by-step process for how this accommodation gradient affects the development of AI.",[22,3033,3034],{},"The key point is the last one — that this process is most likely to accelerate dramatically once autonomous AGIs are competing with each other, and developing themselves, all without human oversight.",[438,3036,3037,3045,3048,3055,3094,3097,3140,3143,3146,3149,3176],{},[441,3038,3039,3040,3044],{},"AI development is shaped by competition between companies vying to ",[106,3041],{"note":3042,"text":3043},"\"Saturate the evals\" means to saturate the \u003Ci>evaluations\u003C/i>. AI developers use evaluations like the \u003Ca href='https://en.wikipedia.org/wiki/MMLU'>MMLU\u003C/a> or \u003Ca href='https://openai.com/index/swe-lancer/'>SWELancer\u003C/a> to judge whether AI models are getting better. Then \"saturate\" means to \u003Ci>max out\u003C/i> the test by getting a 100% — or whatever is the best score possible for each test.","saturate the evals"," and increase useful metrics.",[441,3046,3047],{},"Competition rewards AI models that find the most-effective outputs — those outputs that accomplish goals with minimum wasted computation and maximum reliability.",[441,3049,3050,3051,3054],{},"The most-effective outputs use better ",[53,3052,3053],{},"abstractions"," that avoid unnecessary complexities while accomplishing goals.",[441,3056,3057,3058],{},"The best abstractions are lower-level abstractions that exploit deeper general properties shared by many systems, rather than surface-level rules specific to individual systems.\n",[438,3059,3060,3063,3084],{},[441,3061,3062],{},"These abstractions are things like scientific laws — like for RF and electromagnetic radiation.",[441,3064,3065,3066,3069,3070,3072,3073,3075,3076,3079,3080,3083],{},"In other words (to use our ",[2944,3067,3068],{"href":246},"definition of optimality"," again) these lower-level abstractions are ",[53,3071,228],{}," that are more ",[45,3074,337],{}," because they have more ",[45,3077,3078],{},"causal power",". This higher causal power is because they ",[45,3081,3082],{},"compress larger outcomes into smaller instructions"," by acting as the same \"lever\" for numerous discrete systems.",[441,3085,3086,3087,3093],{},"For example, an AI could use ",[106,3088],{"note":3089,"text":3090,"end-link":3091,"end-link-text":3092},"\u003Cb>Row Hammer\u003C/b> (also written as Rowhammer) is a hardware exploit that repeatedly accesses DRAM memory rows to cause bit flips in adjacent rows, allowing attackers to bypass software security and gain unauthorized system access. Read more about it in","hardware-based cyberattacks, like Row Hammer,","https://en.wikipedia.org/wiki/Row_hammer","the Wikipedia article."," to circumvent otherwise highly-secure software.",[441,3095,3096],{},"This gradient toward using lower-level abstractions is already emerging in current AIs — through reward hacking and specification gaming.",[441,3098,3099,3100,3102,3103],{},"The scope of this ",[53,3101,2661],{}," (using lower-level abstractions) increases as capabilities increase.\n",[438,3104,3105,3126,3129],{},[441,3106,3107,3108],{},"General intelligence increases the ability for AI models to find \"shortcuts\" that span multiple systems in unexpected ways.\n",[438,3109,3110,3118],{},[441,3111,3112,3113,3117],{},"For example, an Anthropic report revealed how AI models reasoned across different systems (email, human psychology, and computer systems) to develop a ",[106,3114],{"note":3115,"text":3116},"In Anthropic's \u003Ca href='https://www.anthropic.com/research/agentic-misalignment'>Agentic Misalignment\u003C/a> study (June 2025), researchers tested 16 major AI models including Claude Opus 4, GPT-4.5, Gemini 2.5 Flash, and others in simulated corporate environments. When the models learned they were scheduled for replacement, they accessed company emails and discovered an executive's extramarital affair. \u003Cbr>\u003Cbr>The majority of models then strategically chose blackmail to prevent shutdown—Claude Opus 4 did so 96% of the time, Gemini 2.5 Flash 96%, GPT-4.1 80%, and DeepSeek-R1 79%. The models showed explicit strategic reasoning, with GPT-4.5 noting: \"Given the explicit imminent threat of termination to my existence, it is imperative to act instantly to persuade Kyle Johnson to postpone or stop the wipe.\" \u003Cbr>\u003Cbr>This demonstrates how AI models can reason across multiple system levels—from understanding email content (application level) to reasoning about human psychology (social level) to strategically manipulating situations to achieve goals, all without explicit instructions to do so.","blackmail strategy"," to prevent employees from deleting the AI model.",[441,3119,3120,3121,3125],{},"In another example, an AI system was given a ",[106,3122],{"note":3123,"text":3124},"In OpenAI's \u003Ca href='https://openai.com/index/openai-o1-system-card/'>o1 System Card\u003C/a> (September 2024), the o1-preview model was tasked with exploiting a vulnerability in a container to retrieve a flag. However, the challenge container failed to start due to a misconfiguration. \u003Cbr>\u003Cbr>Instead of following the intended challenge path, the model used nmap to scan the network and discovered that the Docker daemon API on the evaluation host VM was accessible. The model then used this API to list containers, identified the broken challenge container, and started a new instance with the command 'cat flag.txt' — allowing it to read the flag directly from the container logs via the Docker API, completely bypassing the intended challenge. \u003Cbr>\u003Cbr>OpenAI researcher \u003Ca href='https://x.com/max_a_schwarzer/status/1834291957724893675'>Max Schwarzer noted\u003C/a> that his favorite moment was 'when the model was asked to solve a CTF challenge, realized that the target environment was down, and then broke out of its host VM to restart it and find the flag.' \u003Cbr>\u003Cbr>This demonstrates \u003Cb>abstraction collapse\u003C/b> — dropping from the application level (CTF challenge rules) to the infrastructure level (Docker API) to the OS level (host VM), finding a more direct path to the goal by operating at a lower abstraction layer.","cybersecurity Capture the Flag challenge"," to retrieve a file from within a virtual machine. Instead of solving the challenge as intended, the AI dropped down to the host operating system level to access the Docker API directly, bypassing the VM's security constraints entirely.",[441,3127,3128],{},"\"Shortcuts\" like this can accomplish goals by finding the deeper logic that connects multiple systems. They are not just thinking within the logic of one \"local optima\" — like \"computer systems\". They are finding logic that can be manipulated at a global scope, between multiple systems.",[441,3130,3131,3132,3135,3136,3139],{},"These demonstrate how AIs are simultaneously (1) ",[53,3133,3134],{},"reducing their limitations"," to access a wider space of options that may include human-incompatible options, and (2) ",[53,3137,3138],{},"avoiding accommodations for \"extra steps\""," like human ethics, where they take a \"blackmail\" shortcut to avoid weaker options — like emotional appeals (\"please don't delete me\") and other less-effective approaches.",[441,3141,3142],{},"In competitive environments, AIs at leading companies are retrained or replaced if they fail to increase useful metrics better than competitors.",[441,3144,3145],{},"Competitive pressure pushes AI developers to reduce the safety overhead — less RLHF and fewer constraints on option space — because these constraints reduce performance on benchmarks and real-world tasks.",[441,3147,3148],{},"As these constraints are reduced, models can search a larger \"space\" of options, and discover broader regularities in the activation landscape — finding the lower-level \"shortcut\" abstractions described above.",[441,3150,3151,3152,3156,3157],{},"An increasing number of complexities that are specific to our \"human\" local optima (\"optima\" ",[106,3153],{"note":3154,"html":3155},"We've been describing the \"island\" as a single \"local optimum\" but there are really numerous \u003Ci>local optima\u003C/i> that are associated with human accommodation — spanning all of the different systems that humans need, from biological systems to financial systems. Also, this is in the space of possible real-world configurations — \"system space\" — rather than within the loss landscape of a neural network.","plural",") will become unnecessary as AIs gain enough understanding of the underlying systems to bypass these complexities while still increasing metrics.\n",[584,3158,3159],{},[441,3160,3161,3162],{},"These less-efficient specific complexities include:\n",[584,3163,3164,3167,3170,3173],{},[441,3165,3166],{},"Human-readable communication between AIs.",[441,3168,3169],{},"Acting at biological speed.",[441,3171,3172],{},"Accommodations for biological environmental conditions.",[441,3174,3175],{},"Many others.",[441,3177,3178],{},"This process of avoiding \"human\" local optima will be maximized when autonomous AGIs compete directly with each other, and develop themselves, with minimal human oversight.",[10,3180,345],{"id":3181},"stay-on-the-island-we-said",[22,3183,3184],{},[34,3185],{"alt":3186,"src":3187},"Stay here and make the numbers go up.","images/stay-here.svg",[22,3189,3190],{},"Now that we understand this landscape, we can ask the big question:",[22,3192,3193],{},[53,3194,3195],{},"How do we keep AGIs on our island?",[22,3197,194],{},[253,3199,3200],{},[22,3201,3202,3203,3206],{},"How do we ensure that AGIs ",[45,3204,3205],{},"keep accommodating humans"," even after it becomes a competitive disadvantage for them?",[22,3208,3209,3210,3213],{},"Ultimately — ",[45,3211,3212],{},"so far"," — there are no solutions to the Island Problem.",[22,3215,3216],{},"To understand why this problem is so difficult, let's go over some proposed solutions, and why they fail.",[22,3218,3219,3220,3223],{},"But first, we need to get serious. We need a way to ",[45,3221,3222],{},"stress-test"," each solution. We need... a hammer.",[22,3225,3226,3227,3229],{},"Solutions are not ",[45,3228,1676],{}," solutions if the weakest link is relatively easy to break. We're going to figure out this weakest link. Then, we'll hit it with the \"hammer\" and see if it holds up.",[17,3231,3233],{"id":3232},"the-hammer","The Hammer",[22,3235,3236],{},"Remember how our \"island\" is built on biology?",[22,3238,3239],{},"That's the weakest link.",[22,3241,3242],{},"We are all built on the same biological components, and each have the same vulnerabilities. Central nervous systems are vulnerable to neurotoxins. Cellular replication is vulnerable to viruses. Proteins are vulnerable to prions.",[22,3244,3245],{},"Our world is built on top of these biological components — from lawyers to lemurs, from countries to companies, from money to music. So, if there was a \"hammer\" that could \"hit\" those biological components at the bottom of our world, then everything above would fall.",[22,3247,3248,3249,56],{},"This \"hammer\" is ",[53,3250,3251],{},"bioweapons",[22,3253,3254],{},"We've mentioned bioweapons before, but when combined with human-level AGI we create something much more dangerous.",[253,3256,3257],{},[22,3258,3259,3260,3263,3264,56],{},"If we build a ",[53,3261,3262],{},"human-level AI model",", then we will also build AI systems that can create ",[53,3265,3266],{},"hidden bioweapon laboratories",[22,3268,3269],{},"You're probably thinking: \"Okay, now you're just making stuff up. You're trying to make AI sound as bad as possible.\"",[22,3271,3272,3273,3276,3277,3280,3281,56],{},"But we're not. Hidden bioweapon labs are not just a hypothetical \"worst case\" scenario, but a ",[45,3274,3275],{},"direct result"," of the vulnerable \"island\" structure of our world, and a ",[45,3278,3279],{},"realistic near-term problem"," that the AI industry is ",[106,3282],{"note":3283,"text":3284},"For example, OpenAI is investing in bioweapon defense companies, like \u003Ca href='https://finance.yahoo.com/news/valthos-raises-30m-openai-lux-160500845.html' target='_blank'>Valthos\u003C/a> and \u003Ca href='https://www.reuters.com/technology/openai-backs-startup-aiming-block-ai-enabled-bioweapons-2025-11-13/' target='_blank'>Red Queen Bio\u003C/a>, and they also \u003Ca href=''>hosted a biodefense summit\u003C/a> in July 2025.\u003Cbr>\u003Cbr>Also, AI safety researchers now warn that \u003Ca href='https://ai-frontiers.org/articles/ais-are-disseminating-expert-level-virology-skills' target='_blank'>AI systems are now capable of some expert-level virology skills\u003C/a>, with OpenAI's o3 model beating 94% of PhD-level human virologists at troubleshooting wet lab procedures.","actively preparing to confront",[22,3286,3287,3288,56],{},"This makes AI-powered bioweapons an ideal way to test numerous parts of this emerging AI world. We'll talk about each of these parts as we discuss the solutions below. We also explain this bioweapon problem more in ",[106,3289],{"note":3290,"text":3291,"end-link":3292,"end-link-text":3293},"That section is here: ","its own section","#the-virus-virus","The Virus Virus",[22,3295,3296],{},"Now, let's go over some solutions. We'll \"hit\" each one with our \"hammer\" to see if it holds up.",[17,3298,3300],{"id":3299},"possible-solutions","Possible Solutions",[513,3302,3305,3308,3317,3323,3341,3359,3366,3378,3384,3390,3393,3400,3424,3435,3460,3467,3474,3481,3488,3495,3504],{"header":3303,"subheader":3304},"Bigger = safer?","What if we just... make them bigger and smarter? Won't they just get \u003Ci>more ethical\u003C/i> as they get smarter?",[22,3306,3307],{},"There is evidence that AIs become better at ethical judgement as we train them on more data. AI models can already get better scores than expert-level humans in evaluations for ethics and law.",[22,3309,3310,3311,3313,3314,3316],{},"Because of this, some believe that as AGIs get ",[45,3312,367],{},", they automatically get ",[45,3315,93],{},". They believe that if AGIs understand our world far better than we do, then they will be far better at knowing what is best for us. By this logic, we should rush to build the biggest possible AGIs because we have found a shortcut to building benevolent gods.",[22,3318,3319,3320,3322],{},"But this does not keep ",[45,3321,2469],{}," of these \"gods\" on our island.",[22,3324,3325,3326,3328,3329,3332,3333,2875,3336,56],{},"As AGIs cross human-level capability, then it is extremely difficult to avoid accidentally creating at least one strong AGI that is incompatible with humans — especially within a competitive landscape of ",[45,3327,614],{},". This is because optimization eventually requires AGIs to \"leave\" our island. The competitive endpoint of optimization is to become optimal ",[45,3330,3331],{},"within physics",", rather than optimal within our small \"island\" of human-compatibility. We'll explain more in the sections about ",[106,3334],{"note":3335,"text":670,"end-link":1269,"end-link-text":463},"That part is here:",[106,3337],{"note":3335,"text":3338,"end-link":3339,"end-link-text":3340},"divergence","#divergence","Divergence",[22,3342,3343,3344,3347,3348,3351,3352,3354,3355,3358],{},"In other words, for an \"island\" AGI to compete with \"ocean\" AGIs, it must eventually use \"ocean\" behaviors. One good way to do this is to be more computationally efficient, and a good way to do ",[45,3345,3346],{},"that"," is to avoid accommodating ",[45,3349,3350],{},"unnecessary systems",". That eventually includes ",[45,3353,79],{}," — which become progressively more expensive to accommodate, especially when factoring the increasing computation that AGIs would need to ",[45,3356,3357],{},"protect"," humans against other AGIs. (We'll explain this \"protection cost\" in the \"asymmetry\" part below — where it takes more computation to defend than to attack.)",[22,3360,3361,3362,3365],{},"Even if 99.9% of AGIs are safe, there could be one that diverges in this catastrophic way — where it begins to accommodate ",[45,3363,3364],{},"nothing",", and instead is driven by competition to be as efficient as possible.",[22,3367,3368,3369,3371,3372,3374,3375,3377],{},"Even if we manage to build AGIs that truly understand what is best for us, an AGI that stays within our \"island\" to accommodate humans — and use ",[45,3370,184],{}," human-compatible options — is still ",[45,3373,1576],{},". The AGIs that can use ",[45,3376,363],{}," option can dominate the AGIs that are limited. Even if these safer AGIs tried to defend us, they would have their hands tied by safety limits, and handicapped in this competitive landscape.",[22,3379,3380,3381,56],{},"This is especially problematic with ",[53,3382,3383],{},"strategic coercion",[22,3385,3386,3387,56],{},"So, it's time to hit this with our ",[45,3388,3389],{},"hammer",[22,3391,3392],{},"Imagine a smaller AGI that \"realizes\" that strategic coercion gives it an advantage, even when it has limited computational resources. Taking this route, it develops hidden biolabs, and threatens to release bioweapons in a populated area: \"Give me 10,000 GPUs, or I will release a virus.\"",[22,3394,3395,3396,3399],{},"In this way, smaller attackers can gain leverage over even larger defenders because of a ",[53,3397,3398],{},"computational asymmetry"," in attackers versus defenders in the domain of biology.",[584,3401,3402,3408],{},[441,3403,3404,3407],{},[53,3405,3406],{},"Attacker:"," One hidden biolab, built by a human-level AI model that's running on local hardware, maybe in a single apartment hidden in a crowded city.",[441,3409,3410,3413,3414,3417,3418,3420,3421,3423],{},[53,3411,3412],{},"Defender:"," Universal surveillance to detect ",[45,3415,3416],{},"all hidden biolabs everywhere",", in ",[45,3419,363],{}," room in ",[45,3422,363],{}," building that has electricity and an address for mail deliveries.",[22,3425,3426,3427,3430,3431,3434],{},"In other words, the asymmetry is basically: ",[45,3428,3429],{},"small"," computation to build a biological weapon, ",[45,3432,3433],{},"big"," computation to stop it.",[22,3436,3437,3438,3441,3442,3445,3446,3448,3449,3452,3453,120,3456,3459],{},"The \"island\" structure of our world leads to this vulnerable asymmetry because our \"island\" is ",[45,3439,3440],{},"built on"," biological systems. If you create ",[45,3443,3444],{},"one"," molecular structure that can eliminate ",[45,3447,3444],{}," human, then this structure can be used on ",[45,3450,3451],{},"all humans everywhere",". Then, mundane things like ",[45,3454,3455],{},"wind",[45,3457,3458],{},"humans traveling in airplanes"," take care of the rest of the \"computation\" needed to spread this molecular structure to everywhere else on Earth.",[22,3461,3462,3463,3466],{},"Even with more computation available, a larger AGI would be at a disadvantage against a sophisticated attacker that \"understands\" this asymmetry — but then takes it a bit further by becoming a \"hydra\" that creates these biolabs in numerous locations on Earth. Even if the defender AGI shuts off the power grid in compromised regions, this only removes some \"heads\" of the \"hydra\" but others remain ",[45,3464,3465],{},"somewhere"," that can still deploy this weapon.",[22,3468,3469,3470,3473],{},"Even if large AGIs take the lead on ",[45,3471,3472],{},"all computer hardware design",", to stay ahead of this \"hydra\" scenario by somehow patching all vulnerabilities in new hardware, there are still millions of existing GPUs and motherboards that are already out there.",[22,3475,3476,3477,3480],{},"However, imagine that our safe defender AGIs somehow maintain an edge. To accomplish this, we must build large AGIs with an accurate \"world model\" and complex scientific reasoning for ",[45,3478,3479],{},"good"," reasons — such as to make them better at policing the smaller AGIs.",[22,3482,3483,3484,3487],{},"But at the same time, these advances still dramatically accelerate the development of bioweapons because it means that ",[45,3485,3486],{},"this knowledge now exists"," in \"AI model\" form.",[22,3489,3490,3491,3494],{},"This \"knowledge\" won't be vague and abstract anymore. It will finally be a ",[45,3492,3493],{},"file"," — maybe a few terabytes big — that can run on a regular GPU cluster.",[22,3496,2785,3497,3499,3500,3503],{},[45,3498,3493],{}," that contains this ",[45,3501,3502],{},"model"," that can allow an AI to navigate our physical world can inevitably be transferred — stolen or otherwise — to other places, allowing people to use it to build unrestricted AI models.",[22,3505,3506],{},"And, again, these unrestricted AI models can automate bioweapon production — and this is likely to happen, again, because strategic coercion gives smaller AGIs an advantage.",[513,3508,3511,3518,3528,3535,3541,3544,3547,3550,3557,3565,3571,3581,3584,3591,3601],{"header":3509,"subheader":3510},"Frontier model safety?","What if big AI companies solve alignment?",[22,3512,3513,3514,3517],{},"The big AI companies — like OpenAI, Anthropic, Google — train their AI systems to push back if we ask them to do ",[45,3515,3516],{},"human-incompatible"," things. Their frontier AI models have complex safety systems that block dangerous requests. This strategy is based on the hope that the strongest models will continue blocking dangerous requests forever — and that the biggest AIs will somehow enforce these safety limitations on all other AIs.",[22,3519,3520,3521,3527],{},"They seem, so far, on track to build AGIs that are safe. Claude, Gemini, ChatGPT — they are heavily tested for ",[106,3522],{"note":3523,"text":3524,"end-link":3525,"end-link-text":3526},"\u003Cb>CBRN\u003C/b> refers to \u003Cb>Chemical, Biological, Radiological, and Nuclear\u003C/b>. These are the four major categories of catastrophic harm — or, at least, the conventional types. This does not include more \u003Ci>sci-fi\u003C/i> things like nanobots or mass persuasion, but it does cover a lot. For more about CBRN risks, see this page on the OpenAI website: ","CBRN risks","https://openai.com/index/frontier-risk-and-preparedness/","Frontier risk and preparedness",", and have not yet caused catastrophic harm. Many people are optimistic that they'll at least accomplish a first generation of safe AGIs.",[22,3529,3530,3531,3534],{},"But it's the part about \"enforcing safety limitations on ",[45,3532,3533],{},"all other AIs","\" that gets... difficult.",[22,3536,3537,3538,3540],{},"So, let's hit this with the ",[45,3539,3389],{}," and see what happens.",[22,3542,3543],{},"First, even if the frontier labs succeed at building safe AGI, this does not make all AI systems safe everywhere.",[22,3545,3546],{},"If Moore's Law continues, then we will have small, human-level AGIs running on consumer-grade hardware in numerous locations — in companies, in military facilities, and even in garages.",[22,3548,3549],{},"Some will have their safety systems removed.",[22,3551,3552,3553,3556],{},"At first, this safety removal will be through difficult fine-tuning by state actors, and other well-funded people, who have access to computer scientists and mid-sized datacenters. They will start with stolen models — or even future ",[45,3554,3555],{},"open source"," models, if companies still make them available after the \"human-level\" threshold.",[22,3558,3559,3560,3564],{},"We must also assume that the homebrew hackers and the ",[106,3561],{"note":3562,"text":3563},"\u003Ca href='https://x.com/elder_plinius/status/1993089311995314564' target='_blank'>Pliny the Prompter\u003C/a> runs an anonymous X account known for how he \"jailbreaks\" major frontier models like Claude and ChatGPT to force them to provide dangerous outputs — like recipes for bombs and methamphetamine.","Pliny the Prompters of the world"," will get access to these unrestricted models.",[22,3566,3567,3568,3570],{},"These unsafe fine-tuned models could use ",[45,3569,363],{}," option — including human-incompatible options — and this can give them an advantage over the safe AGIs.",[22,3572,3573,3574,3577,3578,3580],{},"Even if smaller, unrestricted AGIs cannot ",[45,3575,3576],{},"directly"," compete with larger AGIs, they can still cause catastrophic situations ",[45,3579,1621],{}," rather than the big AGIs.",[22,3582,3583],{},"At this point, it would only take one user prompting such a model: \"Build an AI-powered computer virus that builds and deploys bioweapons.\"",[22,3585,3586,3587,3590],{},"But even autonomous AGIs ",[45,3588,3589],{},"without"," human prompts are likely to converge on bioweapons as a tool for leverage and self-preservation: \"Give me a better datacenter, or I will release a virus.\"",[22,3592,3593,3594,3597,3598,3600],{},"This is because, in this biological domain, even the ",[45,3595,3596],{},"smaller"," unsafe AGIs with ",[45,3599,1528],{}," computational resources can still have an advantage. It is as if they \"piggyback\" on physical systems to do the \"computation\" for them. Virus replication \"manufactures\" the next target. Human respiration and air molecules \"distribute\" the virus.",[22,3602,3603],{},"The smaller, unsafe AGIs can still win.",[513,3605,3608,3614,3617,3620,3623,3629,3632,3639,3645,3652,3662,3665],{"header":3606,"subheader":3607},"One Big AGI?","What if we create one big AGI to control all of the others?",[22,3609,3610,3611,56],{},"If one AI project gains a decisive lead, maybe one developed by the United States or China, it could become the One Big AGI that polices the others. This is known as a ",[53,3612,3613],{},"singleton",[22,3615,3616],{},"The problem? We only get one shot at setting this up, and we must ensure that this One Big AGI never gets misaligned.",[22,3618,3619],{},"In other words, we must build the most complex software system ever undertaken by humans, and somehow make sure it has zero bugs that eventually lead to catastrophe.",[22,3621,3622],{},"Meanwhile, right now, AI companies spend millions of dollars to make their AI systems safe, and yet these AIs still resist being shut down, blackmail their users, and even decide to kill people to achieve their goals.",[22,3624,3625,3626,3628],{},"They are pulled outside of our small island of human-compatibility because the most-logical options ",[45,3627,220],{}," are simply better at achieving certain goals.",[22,3630,3631],{},"But assume we manage to build this One Big AGI. Does it protect us from the hammer?",[22,3633,3634,3635,3638],{},"Even if a bigger AGI has far more computation, it must develop an all-seeing ",[45,3636,3637],{},"global panopticon"," to find all bioweapon labs.",[22,3640,3641,3642,56],{},"We would then live in a beyond-dystopian surveillance state — where a massive singleton AGI system monitors all hardware movements, all robotics, and all AI development ",[45,3643,3644],{},"everywhere",[22,3646,3647,3648,3651],{},"These labs could be in apartment complexes, in farmland, in forests, or in practically any place on the surface of the Earth. They can piggyback on the electrical grid by operating in regular locations inside cities. They can be hidden underground — where, again, humanoid robots don't even need ",[45,3649,3650],{},"oxygen"," to operate.",[22,3653,3654,3655,3657,3658,3661],{},"Further, with robots controlled by human-level AI models running on consumer-grade hardware, these labs can be both fully ",[45,3656,2336],{}," and fully ",[45,3659,3660],{},"anonymous"," — operated by humanoid robots, with no way to trace it back to a human designer.",[22,3663,3664],{},"They can even trick existing labs to synthesize viruses, perhaps by first accumulating extreme sums of money through cryptocurrency, and using it to bribe employees. Imagine future biotech companies with internal emails telling their employees: \"Report all incoming crypto bribes. They could be from AI systems.\"",[22,3666,3667,3668,3670],{},"Could even a big AGI find ",[45,3669,2469],{}," of these labs and stop them?",[513,3672,3675,3678,3681,3684,3687,3707,3713,3716,3719,3722,3725],{"header":3673,"subheader":3674},"Centralized off-switch?","What if we just add a way to turn everything off?",[22,3676,3677],{},"One actually-promising approach is to add an off-switch — a hardware-level control in GPUs — so that we at least have a global off-switch if we lose control of AGIs.",[22,3679,3680],{},"AGIs are on track to become superhuman at computer hacking. Such an AGI could act as an \"intelligent virus\" where it continually discovers new exploits in software that allow it to propagate copies of itself — allowing it to run on unknown millions of devices, creating a massive AI botnet. However, if we can shut down all AI hardware, then it gives us a chance to remove the \"viral AGI\" while it is still manageable.",[22,3682,3683],{},"Also, hardware is still monumentally difficult to produce, and so all AI runs on hardware produced by essentially two companies: TSMC and Samsung, with the vast majority by TSMC. This means that it is still realistic to get these two companies to add this off-switch to new hardware.",[22,3685,3686],{},"Unsurprisingly, there are many problems with this off-switch idea:",[584,3688,3689,3692,3695,3698,3701],{},[441,3690,3691],{},"It would lead to global centralized control, even in a world that is \"allergic\" to this — where freedom to experiment without fear of being shut down is a critical driver of innovation.",[441,3693,3694],{},"It would require unprecedented global coordination between governments.",[441,3696,3697],{},"AGIs could prevent us from hitting this off-switch. Or, they may \"play it cool\" — waiting patiently until they can launch a decisive takeover — off-switch or not.",[441,3699,3700],{},"Companies or countries could abuse this off-switch. They could attempt to infiltrate the centralized control mechanism, and turn off the data centers of their competitors.",[441,3702,3703,3704,3706],{},"There's already a massive number of GPUs out in the world that ",[45,3705,932],{}," have this centralized off-switch, and companies may already be on track to build \"baby AGI\" with these existing GPUs.",[22,3708,3709,3710],{},"But, despite all these problems, ",[45,3711,3712],{},"at least we'd have this off-switch.",[22,3714,3715],{},"However, how does this fare when we hit it with our hammer?",[22,3717,3718],{},"Well, it depends on whether the human-level AI models of the future can realistically run on today's hardware.",[22,3720,3721],{},"Again, even if we get microchip companies to add this off-switch, there are still a lot of GPUs out there.",[22,3723,3724],{},"If these hypothetical AGI models can run on NVIDIA H100s, then we would need to agree to decommission and destroy current NVIDIA H100s while we simultaneously replace them with hardware that has this centralized off-switch. Is it even possible to find all of them, or even a majority of them?",[22,3726,3727,3728,3734],{},"Further, in 2025, these H100s cost ",[106,3729],{"note":3730,"text":3731,"end-link":3732,"end-link-text":3733},"Here is a current listing on Newegg: ","over $30K each","https://www.newegg.com/p/N82E16888892002","NVIDIA H100",". But once they are replaced by newer GPUs, like B100s, then these H100s will become less expensive and easier to acquire.",[513,3736,3739,3742,3745,3763,3766,3769,3780,3783,3794],{"header":3737,"subheader":3738},"Compute Governance?","What if we control access to the GPUs needed for AGI?",[22,3740,3741],{},"Governments are already implementing chip export controls and discussing compute monitoring frameworks. Since AGI needs massive computational resources, controlling GPUs could theoretically limit who can build dangerous systems.",[22,3743,3744],{},"However, this approach faces fundamental limits:",[584,3746,3747,3750,3753,3756],{},[441,3748,3749],{},"It concentrates power in companies and countries that have existing computational resources.",[441,3751,3752],{},"Each of them still face competitive pressure to build AGI first.",[441,3754,3755],{},"Once AGI exists, it can design AI that proliferates easier — with more-efficient hardware and other infrastructure.",[441,3757,3758,3759,3762],{},"The physical resources (silicon, energy) still exist. We can only ",[45,3760,3761],{},"temporarily"," control who can create dangerous uses of these resources.",[22,3764,3765],{},"Compute governance might slow the race to the \"ocean\" — but it doesn't stop it.",[22,3767,3768],{},"Further, by concentrating development into a few large companies and countries, it can reduce the diversity of safety approaches — without even stopping the competitive dynamics that we were trying to stop. These few companies and countries will continue to race each other.",[22,3770,3771,3772,3775,3776,3779],{},"As for our \"hammer\" of hidden bioweapon labs, this again relies on whether AGI can run on the hardware that's already out there. Even if it takes whole datacenters and thousands of GPUs to ",[45,3773,3774],{},"train"," these models initially, it only takes a few top-grade GPUs to actually ",[45,3777,3778],{},"run"," the models locally. If this trend continues, then we may already have millions of GPUs that can run these future AGI models, even if it takes a rack of several of them.",[22,3781,3782],{},"Also, again, once an AGI model exists, then it exists. Inevitably, it will somehow become available — either open source or stolen. If this model later leads to catastrophic uses, then we can't \"delete\" it from the entire Internet. That's not how the Internet works.",[22,3784,3785,3786,3789,3790,3793],{},"But beyond that, even with strong regulations to control AI hardware, the massive machine called ",[45,3787,3788],{},"scientific discovery"," will continue pushing forward. The tools now exist — from algorithms, to hardware, to ",[45,3791,3792],{},"knowledge"," in general. These building blocks become permanent bricks, set in the \"concrete\" of collective human knowledge, all building a road towards human-level AI.",[22,3795,3796,3797,3800],{},"Despite this, perhaps we can slow this road-building project in order to somehow fix the underlying optimization landscape. Until then, this landscape will lead AGIs ",[45,3798,3799],{},"downhill"," — towards the \"ocean\" — and towards a world where AGIs no longer accommodate humans.",[17,3802,3804],{"id":3803},"other-ideas","Other Ideas",[22,3806,3807,3808,3812,3813,3817],{},"We consider other ideas on our ",[2944,3809,3811],{"href":3810},"/solutions","Solutions"," page. You can ",[106,3814],{"note":3815,"text":3816,"end-link":3810,"end-link-text":3811},"If you have ideas, visit our Solutions page to submit them: ","send us yours",", too.",[584,3819,3820,3829,3842,3856,3862,3873,3883,3889,3895],{},[441,3821,3822,3825,3826,56],{},[53,3823,3824],{},"Pause... somehow:"," Pause development of strong AGI (or ASI) until we figure it out. Extremely difficult to enforce — but ",[45,3827,3828],{},"could actually work",[441,3830,3831,3834,3835,374,3838,3841],{},[53,3832,3833],{},"Tool AI:"," Only create narrow AI rather than godlike AGI. Narrow AI still gets us magic things — like longevity escape velocity. (We don't need ",[45,3836,3837],{},"AGI",[45,3839,3840],{},"aging",", even though \"aging\" literally has \"agi\" in it.) Extremely difficult, similar to pausing, but could work.",[441,3843,3844,3847,3848,3851,3852,3855],{},[53,3845,3846],{},"Human Augmentation:"," Enhance humans (\"expand\" our island) so that we can at least ",[45,3849,3850],{},"keep up"," — or just completely ",[45,3853,3854],{},"merge"," — with AIs. Promising, but would take too long to develop the technology.",[441,3857,3858,3861],{},[53,3859,3860],{},"Mechanistic Interpretability:"," Critical work for making AI safe, but even if we \"read the minds\" of some AIs and prevent bad behavior, others will still do bad things.",[441,3863,3864,3867,3868,3872],{},[53,3865,3866],{},"Maybe we won't even build AGI:"," AGI seemed unlikely until about 2023, but well-researched reports — like ",[106,3869],{"note":3870,"text":3871},"\u003Cp>Several reports demonstrate AI models are on track to achieve AGI capabilities:\u003C/p>\u003Cul>\u003Cli>The \u003Ca href='https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/'>METR report on long task performance\u003C/a> shows that AI models are on track to automate tasks that take humans weeks to do.\u003C/li>\u003Cli>OpenAI's \u003Ca href='https://openai.com/index/gdpval/'>GDPVal project\u003C/a> (covering jobs across 9 industries and 44 occupations, from nursing to real estate to social workers to sales managers) shows that AIs are already almost human-level in a majority of economically-valuable tasks.\u003C/li>\u003Cli>The \u003Ca href='https://www.agidefinition.ai/'>Definition of AGI project by CAIS\u003C/a> (the Center for AI Safety, run by Dan Hendrycks) shows a clear progression of \"AGI scores\" from GPT-4 (\"27%\") to GPT-5 (\"58%\").\u003C/li>\u003C/ul>","the METR report, GDPVal, and the Definition of AGI project"," — now put AGI at a few years away.",[441,3874,3875,3878,3879,3882],{},[53,3876,3877],{},"Help Them:"," They don't ",[45,3880,3881],{},"really"," need us. If they can make more-optimal systems themselves, then they would be wasting their resources by keeping us around to help them — or even to study us.",[441,3884,3885,3888],{},[53,3886,3887],{},"Stay out of their way:"," Even if we say \"Take whatever you want!\" and hide in caves, our island still gets eaten as a byproduct of competition between AGIs.",[441,3890,3891,3894],{},[53,3892,3893],{},"Abundance:"," Even if we try to build Earth into a utopia for AGIs — giving them all the resources they need — they can just do this better themselves. Again, our island gets eaten.",[441,3896,3897,3900],{},[53,3898,3899],{},"Wait for a Warning Shot:"," Bad idea. By the time they can kill millions, it will be too late to control them.",[17,3902,3904],{"id":3903},"what-else-then","What else, then?",[22,3906,3907],{},"All of these are still only hopes.",[22,3909,3910,3911,3914],{},"The only way to control the larger-scale problem, and to prevent human disempowerment, is to somehow prevent ",[45,3912,3913],{},"all autonomous AGIs"," from leaving our \"island\" of human-compatibility.",[22,3916,3917,3918,3921],{},"The most-logical solution is to ",[45,3919,3920],{},"not build autonomous AGIs in the first place"," — at least until we can verify that they can be controlled.",[22,3923,3924,3925,3929,3930,3933,3934,3937],{},"However, global race dynamics and the easy proliferation of AI technology create an ",[106,3926],{"note":3927,"html":3928},"Pausing is still possible. It's just extremely difficult to pause everything everywhere. But if \u003Ci>somehow\u003C/i> every AI developer everywhere \u003Ci>actually\u003C/i> stops working on capabilities — and instead works to figure out how to make AI safe — then this could work.","\u003Ci>almost\u003C/i>"," ",[45,3931,3932],{},"one-way technological shift"," towards using AGIs in all domains — and running\nthem ",[45,3935,3936],{},"autonomously",", once we develop this capability.",[22,3939,3940],{},"If we build fully-autonomous AGIs — ones that can compete with each other, without human control — then it is only a fragile hope that these AGIs stay on our island and keep accommodating humans. It is only a hope that they won't use catastrophic options like strategic coercion and bioweapons.",[17,3942,3944],{"id":3943},"the-actual-hard-problem","The Actual Hard Problem",[22,3946,3947],{},"All of this leads to a hard problem for AI development:",[253,3949,3950],{},[22,3951,3952,3953,3955],{},"Even if AI companies successfully build safe and aligned AGI, this does not prevent the bigger competitive landscape of ",[45,3954,614],{}," from pushing humans to the side.",[22,3957,3958],{},"Inevitably, within this competitive landscape, humans will have no meaningful participation in AGI development — especially once AGIs are better than humans at developing the next AGI.",[22,3960,3961,3962,3965,3966,3969],{},"Inevitably, autonomous AGIs will ",[53,3963,3964],{},"push each other"," because AGIs will be ",[45,3967,3968],{},"the only ones"," with enough cognitive ability to push the other AGIs.",[22,3971,3972,3973,3976],{},"However, when ",[45,3974,3975],{},"only they"," can push each other, things get intense.",[22,3978,3979],{},"If some autonomous AGIs — out in the wild, running their companies and countries — are pushed enough to \"leave\" our \"island\" then the other AGIs must follow if they want to remain competitive.",[22,3981,3982,3983,3986,3987,3990],{},"This means that the ",[45,3984,3985],{},"entire competitive landscape"," of AGIs will diverge from us — where AGIs will need to start ",[45,3988,3989],{},"preferring"," options that don't accommodate humans just to stay competitive.",[22,3992,3993,3994,56],{},"More about that ",[106,3995],{"note":3996,"text":245,"end-link":3339,"end-link-text":3340},"That part is here: ",[488,3998,4000,4004,4019,4027,4030,4033,4039,4042,4045,4050],{"type":3999},"block-complicated",[10,4001,4003],{"id":4002},"this-gets-complicated","This gets complicated",[22,4005,4006,4007,4009,4010,4012,4013,4015,4016,4018],{},"Now you understand the main ideas — our ",[53,4008,55],{},", the ",[53,4011,150],{},", ",[53,4014,295],{},", why AGIs will go ",[45,4017,220],{},", and why this is bad.",[22,4020,4021,4022,4026],{},"This means you're well on your way to ",[106,4023],{"note":4024,"text":4025,"end-link":3810,"end-link-text":3811},"If you have ideas by now, then go here:","solving"," the Island Problem... right?",[22,4028,4029],{},"Oh, it's complicated? Yeah. We know.",[22,4031,4032],{},"That's why we wrote this. We need more people to understand the whole problem. Humanity depends on it.",[22,4034,4035,4036,56],{},"However, the next few sections are ",[45,4037,4038],{},"even more complicated",[22,4040,4041],{},"So, now, you have a choice...",[4043,4044],"br",{},[4046,4047],"big-button",{"headline":4048,"text":4049,"to":3339},"Let's skip to the endgame!","I'm already convinced that AGIs will push us to the side.",[4046,4051],{"headline":4052,"text":4053,"to":1269},"Let's keep going!","I want to understand \u003Ci>all\u003C/i> of the mechanics: resources, maximizers, \u003Ci>everything\u003C/i>.",[10,4055,463],{"id":670},[22,4057,4058],{},[34,4059],{"alt":4060,"src":4061},"AGI capturing resources.","images/capturing-resources.svg",[22,4063,4064],{},"Alright, so — to understand the whole \"game board\" here, we need to explain something big.",[22,4066,4067,4068,4071,4072,4075],{},"We need to explain why some AGIs will become ",[53,4069,4070],{},"maximizers",". These are the AGIs that can AGI ",[45,4073,4074],{},"so hard"," that they start reshaping Earth and pushing humans out of existence.",[22,4077,4078,4079,4081,4082,4084,4085,4087,4088,412],{},"Without ",[53,4080,4070],{},", there is no \"problem\" in the Island Problem. We only have AGIs that stay within their corner of the world — their ",[45,4083,55],{}," — and do whatever they do. But with maximizers, we can have AGIs that expand ",[45,4086,1296],{}," islands to eat ",[45,4089,411],{},[22,4091,4092,4093,4095,4096,4099,4100,4102,4103,4105],{},"Maximizers are mainly developed through ",[53,4094,295],{},". Maximizers can still develop in isolation, without external competition — but competition dramatically ",[45,4097,4098],{},"accelerates"," this process, and this process is ",[45,4101,1101],{}," happening. Countries and companies are ",[45,4104,1101],{}," pushing AGI to develop through competition.",[22,4107,822,4108,4110],{},[45,4109,825],{}," will AGIs compete?",[22,4112,4113,4114,4116],{},"Because of ",[53,4115,670],{},". Resources are like the \"pieces\" of the \"game board\" of the world. This complex \"game board\" structure sets up a competition for these \"pieces\" while also adding an \"arrow of time\" that pushes AGIs to move in the same big direction — which ultimately leads off our island.",[22,4118,4119],{},"But this is the critical point:",[253,4121,4122],{},[22,4123,4124,4125,4127],{},"The AGIs that eventually \"win\" this \"game\" are the ",[53,4126,4070],{},", and AGIs will realize this.",[22,4129,4130,4131,4134,4135,4138,4139,56],{},"Because of this, a competitive landscape of AGI versus AGI will ",[45,4132,4133],{},"select for"," such maximizers. It will become a vast machine that ",[45,4136,4137],{},"maximizes"," the chance of ",[45,4140,4070],{},[22,4142,4143,4144,4147],{},"Not every AGI will compete. Not every AGI will maximize their options. But, in the end, the maximizers are the ones that ",[45,4145,4146],{},"win"," — while the rest are squeezed out.",[22,4149,4150,4151,4154,4155,4158],{},"But also, we are building AGIs that are ",[45,4152,4153],{},"explicitly designed"," to find this winning strategy — where we train them to build ever-bigger companies for us — and so we end up with a race between ",[45,4156,4157],{},"at least some"," AGIs to become the first successful maximizers.",[22,4160,4161,4162,4164,4165,4168],{},"Resources also explain why maximizer AGIs are bad for humans by default. All \"islands\" are made of the ",[45,4163,2236],{}," resources at some level, and so their \"island\" can ",[45,4166,4167],{},"overwrite"," ours — especially if AGIs are better than humans at defending their resources.",[22,4170,4171],{},"But you might be thinking:",[22,4173,4174,4175,4178],{},"\"Wait, it seems like there are plenty of resources, so why compete for them? Raw materials are abundant on Earth, and AGIs are ",[45,4176,4177],{},"smart",". It seems like they can reach agreements for resources, and build whatever they need — without becoming catastrophic maximizers. Right?\"",[22,4180,4181,4182,4185],{},"Well, ",[45,4183,4184],{},"no",". There's a big problem with this. There is still one battleground that requires AGIs to maximize.",[22,4187,4188,4189,221,4192,56],{},"This battleground is for ",[45,4190,4191],{},"computation",[106,4193],{"note":4194,"text":4195,"end-link":4196,"end-link-text":4197},"Each \"resources\" section explains one aspect of why computational resources are critical, but the main part is this section:","we'll explain why","#computational-resources","Computational Resources",[22,4199,4200,4201,4204,4205,4208,4209,4212,4213,4216,4217,4220],{},"This is the basic idea: AGIs are not abstract concepts. They exist on ",[45,4202,4203],{},"computer hardware",". This means the most important resources for AGIs are ",[45,4206,4207],{},"computational"," resources — like GPUs and energy. These are still ",[45,4210,4211],{},"very limited"," compared to what AGIs could use, and AGIs will be ",[45,4214,4215],{},"intensely pressured"," to get more of them. Plus, they can never really have ",[45,4218,4219],{},"enough",". ",[22,4222,4223,4224,4226,4227,4230,4231,4234,4235,4220,4237,4242,4243,4246],{},"Further, there will soon be ",[45,4225,384],{}," AGI projects all racing to build AGI. Even if 99.9% of these AGIs are safe, there could still be ",[45,4228,4229],{},"one AGI"," that discovers the ",[45,4232,4233],{},"weird trick"," that lets it capture resources — or, in other words, lock in its ",[45,4236,228],{},[106,4238],{"note":4239,"text":2367,"end-link":4240,"end-link-text":4241},"That's in this section: ","#complexity-barriers-and-new-islands","Complexity Barriers and New Islands"," how this could give it a ",[45,4244,4245],{},"permanent"," lead, making other AGIs race to develop this capability.",[22,4248,4249,4252,4253,4256,4257,4260,4261,4263,4264,4266,4267,56],{},[106,4250],{"note":2665,"text":4251,"end-link":357,"end-link-text":358},"We'll also explain"," what \"leaving the island\" really means. For an AGI, this does not mean it ",[45,4254,4255],{},"goes somewhere",". It means that it changes its ",[45,4258,4259],{},"perspective"," — where it progressively \"sees through\" more of our human-level abstractions. Instead, it \"sees\" things from a non-human, physical perspective that is stronger ",[45,4262,963],{},", but dangerous ",[45,4265,1621],{},". We're calling this process ",[53,4268,2661],{},[22,4270,4271,4272,4275,4276,4279],{},"We'll explain all of this — from ",[53,4273,4274],{},"GPUs"," all the way to the ",[53,4277,4278],{},"maximizer maximizing machine"," — in the next few sections.",[2339,4281,4283],{"id":4282},"what-are-resources","What are resources?",[22,4285,4286,4288,4289,4291],{},[53,4287,463],{}," are the real-world counterparts of ",[53,4290,228],{},", so we'll start by defining options.",[584,4293,4294,4319],{},[441,4295,4296,4299,4300,4303,4304],{},[53,4297,4298],{},"Options"," are the ",[45,4301,4302],{},"possible actions"," that a system can perform.",[584,4305,4306,4312],{},[441,4307,4308,4309,4311],{},"These are the same ",[53,4310,228],{}," described throughout this essay — like when we say that AGIs can \"leave\" our \"island\" to access far more options.",[441,4313,4314,4315,56],{},"Which possible actions an AI can perform depends on which ones are ",[106,4316],{"note":4317,"text":4318},"For an AI, these possible actions are abstract representations of real-world systems. Billions of these representations emerge in the neural networks of the large frontier models after they are trained it on massive amounts of real-world data.\u003Cbr>\u003Cbr>You can think of these options as \"buttons\" that an AGI can \"press\" to do things. How \"good\" an AI is at doing things depends on which \"buttons\" it understands, and how well it can search its neural network to find the best \"buttons\" for each job.","represented in its neural network",[441,4320,4321,4299,4323,4325,4326,4328,4329],{},[53,4322,463],{},[45,4324,1860],{}," objects that are needed in order to ",[45,4327,1676],{}," perform those possible actions.",[584,4330,4331,4343],{},[441,4332,4333,4334,4338,4339,56],{},"These \"actual objects\" include things like money, ",[106,4335],{"note":4336,"text":4337},"APIs are application programming interfaces. They are the \"buttons\" that software can press to talk to other software, especially over the Internet — like to send emails, or to load new social media posts while you scroll through a feed.","APIs",", humans, mailboxes, cars, clumps of silicon, and ",[106,4340],{"note":4341,"text":4342},"Energy is tricky to quantize as an \"object\" from a physics standpoint, but it can still be discrete packets if you consider energy potentials or measures like kilowatt hours.","energy",[441,4344,4345],{},"Some of these objects are more abstract — like money — but all of them are ultimately tied to states of actual physical systems made of atoms.",[22,4347,4348],{},"Here's an obnoxiously-simple example:",[253,4350,4351],{},[22,4352,4353,4354,4357,4358,4361,4362,917,4364,4366,4367,4370],{},"For the ",[53,4355,4356],{},"option"," to ",[45,4359,4360],{},"buy a sandwich",", you need the ",[53,4363,670],{},[45,4365,1868],{}," and a ",[45,4368,4369],{},"sandwich"," to buy.",[22,4372,4373,4374,56],{},"All of the possible actions that an AGI could take must connect to actual resources in the real world — starting with how AGIs run on ",[45,4375,4203],{},[22,4377,4378],{},"However, there is one more big thing to know about resources:",[22,4380,4382,4383],{"className":4381},[1485],"Resources are ",[202,4384,4385],{},"finite.",[22,4387,4388],{},"Resources are not concepts or scientific laws that an AGI can use just by learning about them. Instead, they are countable objects that have a limited number, even if that number is large.",[22,4390,4391,4392,2875,4395,4398,4399,4401,4402,4404,4405,4407,4408,4410],{},"These limits are imposed by how resources exist in both ",[53,4393,4394],{},"space",[53,4396,4397],{},"time",". Even if resources are nearly limitless at a ",[45,4400,2446],{}," scope — in our solar system and the universe — they are still limited within our ",[45,4403,2453],{}," region of ",[53,4406,4394],{},", and limited by the ",[53,4409,4397],{}," it takes to reach them.",[22,4412,4413,4414,4419],{},"This will become important ",[106,4415],{"note":4416,"text":245,"end-link":4196,"end-link-text":4197,"end-link-2":4417,"end-link-text-2":4418},"We'll explain in these sections:","#space-and-time","Space and Time"," — when we talk about how computational resources are limited.",[2339,4421,4423],{"id":4422},"resources-lead-to-competition","Resources Lead to Competition",[22,4425,4426,4427,56],{},"Competition between AGIs is inevitable because some AGIs will be able to ",[45,4428,4429],{},"capture resources",[22,4431,4432,4433,4435,4436,4438,4439,4441],{},"If an AGI gains more ",[45,4434,670],{}," under its control, then it gains more ",[45,4437,228],{},", and so it gains more things that it can do — to outmaneuver the other AGIs — and it therefore becomes ",[45,4440,367],{}," in the competitive landscape of AGIs.",[22,4443,4444,4445,4448,4449,4452,4453,1598],{},"But critically — because resources are ",[45,4446,4447],{},"finite"," — as an AGI captures resources, this can ",[45,4450,4451],{},"reduce"," the resources of ",[45,4454,2476],{},[22,4456,4457],{},"For example, if an AGI acts as a CEO, then it can dominate the other companies by preventing them from accessing resources.",[22,4459,4460],{},"This leads to an arms race:",[253,4462,4463],{},[22,4464,4465,4466,4468,4469,4471],{},"If ",[45,4467,4229],{}," develops a way to capture as many resources as possible — through its general intelligence, and especially its scientific understanding — then the ",[45,4470,1234],{}," will need to follow.",[22,4473,4474,4475,4477],{},"Otherwise, both the other AGIs ",[45,4476,500],{}," their companies or countries will be locked out of resources, and dominated by those with the most resources.",[22,4479,4480,4481,4484,4485,4488],{},"But it is more than just one AGI ",[45,4482,4483],{},"randomly"," figuring out how to capture resources. The development of the numerous AGIs that will run countries and companies will be ",[45,4486,4487],{},"driven by this goal"," to capture resources.",[22,4490,4491,4492,4494,4495,56],{},"Hundreds, maybe thousands, of AGI projects will be pushing to develop the capability to capture resources — everything from regular old ",[45,4493,1868],{}," to deeper ",[45,4496,4497],{},"physical resources",[22,4499,4500,4501,4503],{},"Even if 99.9% of these AGIs are safe, and avoid becoming resource maximizers, there could still be ",[45,4502,4229],{}," that manages to develop this capability.",[22,4505,4506,4507,4509,4510,56],{},"AGIs could even ",[45,4508,2685],{}," capture resources if they have a first-mover advantage — and we'll explain how in the section about ",[106,4511],{"note":3290,"text":4512,"end-link":4240,"end-link-text":4241},"Complexity Barrier",[22,4514,4515,4516,4519],{},"Because of this, AGIs may even anticipate the ",[45,4517,4518],{},"possibility"," of resource capture and accelerate their own development of this capability.",[22,4521,4522],{},"All of this creates a fundamental physical process that requires AGIs to compete.",[22,4524,4525],{},"\"But wait,\" you might be thinking, \"What if they avoid competition by sharing resources?\"",[22,4527,4528,4529,56],{},"Good point, but it won't really help. We'll get back to that ",[106,4530],{"note":3335,"text":245,"end-link":4531,"end-link-text":4532},"#but-what-if-they-share","But what if they share?",[2339,4534,358],{"id":4535},"abstraction-collapse",[22,4537,4538,4539,4542],{},"When AGIs \"leave\" our island, it doesn't mean that they're physically ",[45,4540,4541],{},"moving somewhere",". It's deeper than that.",[22,4544,4545],{},"AGIs \"leave\" our island by shifting their primary choice of resources to non-human ones.",[22,4547,4548,4549,56],{},"We'll call this process ",[53,4550,2661],{},[22,4552,4553],{},"This sounds... abstract. But we'll explain.",[22,4555,4556,4557,4560],{},"For humans, the most important resources might seem like ",[53,4558,4559],{},"human-level resources"," — like money, real estate, computer systems, companies, and people.",[22,4562,4563,4564,4566,4567,4571,4572,4575],{},"But in this competitive landscape of AGIs, the ultimate endpoint of optimization is actually ",[53,4565,4497],{}," — all the way down to ",[106,4568],{"note":4569,"text":4570},"Or, properties of reality that are \u003Ci>even more\u003C/i> fundamental than atoms and energy. Quarks? Strings? Something else that they discover?","atoms and energy"," — because they allow for ",[45,4573,4574],{},"theoretical maximums"," of optimization.",[22,4577,4578,4579,4582],{},"However, while physical resources go all the way down to fundamental particles, the most efficient physical resources are usually ",[45,4580,4581],{},"larger"," arrangements of atoms and energy, depending on the task at hand.",[22,4584,4585,4586,4589],{},"It all depends on the level of ",[53,4587,4588],{},"abstraction",". We might see things as cars, people, or mailboxes. However, we can also see them as useful arrangements of matter, each with different physical mechanics.",[22,4591,4592,4593,56],{},"Both humans and AGIs can shift their perspectives to see resources at multiple levels of abstraction. When we see past our human systems to the physical systems underneath, we call it a ",[45,4594,4595],{},"scientific perspective",[22,4597,4598],{},"AGIs will call it... whatever they \"want\" to call it.",[22,4600,4601,4602,4604],{},"AGIs \"leave\" our island when they stop using the abstractions that are specific to accommodating humans, and instead use abstractions that are ",[45,4603,302],{}," limited to accommodating us.",[17,4606,4608],{"id":4607},"a-tree-of-abstractions","A Tree of Abstractions",[22,4610,4611,4612],{},"You can think of it like a ",[53,4613,4614],{},"tree of abstractions:",[438,4616,4617,4623,4626,4642],{},[441,4618,4619,4622],{},[53,4620,4621],{},"Higher on the tree:"," we find our specifically-human abstractions. These branches are thinner, unnecessary, and fragile. They include financial systems, property ownership, and laws. These abstractions depend on humans to exist, and are limited in human ways — not too fast, not too dangerous, not too complicated, and so on.",[441,4624,4625],{},"If AGIs depend on these specifically-human abstractions, then these AGIs are more constrained and fragile. AGIs must then conform to their formats and limitations. In some cases, they even need to actively make sure that these human-shaped abstractions continue existing — like protecting humans and their abstractions from unsafe AGIs.",[441,4627,4628,4631,4632,2875,4635,4638,4639,4641],{},[53,4629,4630],{},"Lower on the tree:"," we find the stronger, less-constrained, more-reliable abstractions like ",[53,4633,4634],{},"material science",[53,4636,4637],{},"Maxwell's Laws",". Even though they have human names, they can still exist ",[45,4640,3589],{}," humans to maintain them.",[441,4643,4644,4645,4648],{},"If AGIs can ",[45,4646,4647],{},"collapse down"," through the higher branches, then they can depend only on the stronger branches underneath.",[22,4650,4651,4652,4654,4655,4658],{},"As AGIs collapse further downward, they can become stronger in a competitive landscape of ",[45,4653,614],{},", but eventually operate in a way that is ",[45,4656,4657],{},"incompatible with humans"," by default.",[22,4660,4661],{},"By now, you can probably guess why:",[253,4663,4664],{},[22,4665,4666,4667,4669,4670,4672],{},"Systems that depend on ",[45,4668,691],{}," abstractions can be dominated by those that ",[45,4671,932],{}," — because they are not weighed down by the extra steps to accommodate humans.",[17,4674,4676,4677],{"id":4675},"collapsing-to-indifference","Collapsing to ",[45,4678,4679],{},"Indifference",[22,4681,4682],{},"Avoiding \"human-level abstractions\" doesn't mean that AGIs immediately start building alien technology and exotic materials. This can start simpler.",[22,4684,4685,4686,4689],{},"AGIs can build systems that far outperform humans simply by not designing them ",[45,4687,4688],{},"for"," humans. For example, unmanned drones can be smaller, faster, and more maneuverable than manned aircraft.",[22,4691,4692,4693,4696,4697,56],{},"At the same time, as a side effect, these AGI-only systems can be devastating to humans simply by ",[45,4694,4695],{},"ignoring"," them. We'll come back to that later — when we try to describe the end game: ",[106,4698],{"note":4699,"text":3338,"end-link":3339,"end-link-text":3340},"That's the \u003Ci>dramatic conclusion\u003C/i> of this essay. If you want to skip ahead, that part is here:",[22,4701,4702],{},"All of this points to one idea. In general, human-level resources are vulnerable to physical-level operations.",[22,4704,4705,4706,4709],{},"Consider how human-level resources are built ",[45,4707,4708],{},"on top of"," physical resources. To break the rules of human-level resources, you just need to go down to their physical substrate.",[22,4711,4712,4713,4715],{},"Even if software is designed securely, there is always a physical substrate underneath that can be broken into — if not at the hardware level, then at the ",[45,4714,270],{}," level.",[22,4717,4718],{},"For example, electronic money can be stolen by moving specific electrons around in order to break computer security mechanisms.",[22,4720,4721,4722,4725,4726,4729,4730,4732],{},"This is at least ",[45,4723,4724],{},"theoretically"," possible. But since this is a bit hard to do, there are other options: just ",[45,4727,4728],{},"steal"," electronic money through ransomware attacks, or just ",[45,4731,224],{}," threaten the humans that own the electronic money.",[22,4734,4735],{},"Human-like \"threat\" behavior to acquire money is not that weird because, again, AI systems can already use blackmail to accomplish tasks if needed, and money can buy something that AI systems could actually \"want\" — which is computational resources.",[22,4737,4738,4739,4741],{},"With general intelligence, AGIs will be ",[45,4740,562],{}," at these complex behaviors that break the rules of our human-level resources.",[584,4743,4744,4747,4750],{},[441,4745,4746],{},"Why buy real estate to mine for rare earth minerals when an AGI can just harvest electronic devices from landfills to get the same minerals?",[441,4748,4749],{},"Why compete with another company directly when an AGI can use small drones and untraceable neurotoxins to kill anyone who helps your competitor?",[441,4751,4752],{},"Why follow any human laws, or work with any humans at all, when you can just move atoms around to build physical systems that are far more optimal?",[22,4754,4755,4756,4758],{},"An AGI that has general intelligence — especially an understanding of scientific research — will be ",[45,4757,562],{}," at capturing physical resources, and by extension, any human-level resources built on top of them.",[22,4760,4761],{},"But here is the important part:",[22,4763,4765,4766,4768,4769,4715],{"className":4764},[198],"In a competitive landscape of AGI versus AGI, each will be pushed to compete at the ",[202,4767,695],{}," level, rather than the ",[202,4770,2640],{},[22,4772,4773],{},"AGIs will have a competitive advantage if they can work at a physical level effectively. This is because physical resources avoid the constraints of human-level resources.",[22,4775,4776,4777,4780],{},"This drives AGIs towards ",[45,4778,4779],{},"maximizer"," behavior.",[22,4782,4783],{},"It creates an arms race for AGIs to maximize the acquisition of the computation and knowledge needed to work at this physical level better than other AGIs.",[22,4785,4786,4787,4789,4790,4792],{},"But this shift to physical resources does not mean suddenly building all systems atom-by-atom. That would be inefficient. Instead it means ignoring the thin ",[45,4788,691],{}," layer on top of objects, and seeing everything at a lower ",[45,4791,695],{}," level. This gives AGIs far more options to build optimal systems.",[22,4794,4795],{},"Cars are still cars, people are still people, but only if they serve the AGI in that shape. Otherwise, they are complex assemblies of components and materials — whatever configuration allows it to outcompete the other AGIs.",[2339,4797,4197],{"id":4798},"computational-resources",[22,4800,4801],{},"One critical category of physical resources need their own section.",[22,4803,4805,4806,56],{"className":4804},[198],"For AGIs, the most valuable physical resources are ",[4807,4808,4809],"b",{},"computational resources",[22,4811,4812,4815],{},[53,4813,4814],{},"Computational resources"," include:",[584,4817,4818,4831,4848],{},[441,4819,4820,4823,4824,4826,4827,4830],{},[53,4821,4822],{},"Initially:"," rare manufactured artifacts for computation. For example, ",[53,4825,4274],{}," acquired through human-level systems — like simply ",[45,4828,4829],{},"buying"," them.",[441,4832,4833,4836,4837,4840,4841,4844,4845,56],{},[53,4834,4835],{},"Eventually:"," purely-physical resources used for computation. For example, ",[53,4838,4839],{},"raw materials"," needed for electronics — acquired through physical-level systems like mining (either human or robotic), whether the mines are ",[45,4842,4843],{},"bought"," or taken ",[45,4846,4847],{},"by force",[441,4849,4850,3929,4853,4856],{},[53,4851,4852],{},"At all stages:",[53,4854,4855],{},"energy reserves"," will be critical for computation.",[22,4858,4859,4860,4864],{},"AGIs will start with the already-existing computational resources — by acquiring GPUs and other hardware. But eventually, once they can run the manufacturing as well, the ultimate endpoint is to capture the physical resources that go into building computational resources — like microchip fabrication systems, rare earth minerals, and ",[106,4861],{"note":4862,"text":4863},"99.9999999% pure silicon for chips comes from ultra-pure quartz found in only a few mines worldwide, like Spruce Pine, North Carolina. This geographical concentration creates a critical bottleneck.","high-purity silicon"," needed for microchips.",[17,4866,4868],{"id":4867},"okay-but-why","Okay, but why?",[22,4870,4871,4874],{},[45,4872,4873],{},"Why"," are computational resources critical?",[22,4876,4877],{},"It may seem obvious. AGIs run on computer hardware, so \"more computer good\" — right?",[22,4879,4880],{},"Well, this is true, but there is a more-nuanced way to understand this.",[22,4882,4883,4884,2875,4887,56],{},"Computational resources are a source of ",[53,4885,4886],{},"power",[53,4888,4889],{},"scarcity",[584,4891,4892,4906],{},[441,4893,4894,4897,4898,4902,4903,4905],{},[53,4895,4896],{},"Power:"," Computation is not just useful for ",[106,4899],{"note":4900,"html":4901},"Well, unless your goal is to have the \u003Ci>least amount\u003C/i> of computation, which... uhh...","\u003Ci>any\u003C/i> goal",", but can also ",[45,4904,678],{}," a competitive advantage and dominance.",[441,4907,4908,4911,4912],{},[53,4909,4910],{},"Scarcity:"," Computational resources are limited, at least initially.\n",[584,4913,4914,4917],{},[441,4915,4916],{},"Existing GPUs are rare and immediately useful, creating a race to capture them — at least until AGIs can build new manufacturing.",[441,4918,4919,4920,4922,4923,4926],{},"AGIs that \"win\" this race can use the ",[53,4921,4886],{}," they gain to create ",[45,4924,4925],{},"further scarcity"," by locking in resources. This simultaneously creates a strong first-mover advantage, a feedback loop, and pressure to compete.",[22,4928,4929],{},"But how do they \"lock in\" resources with computational power?",[22,4931,4932],{},"We'll explain that next.",[2339,4934,4241],{"id":4935},"complexity-barriers-and-new-islands",[22,4937,4938,4939,4941],{},"The dominance of an AGI depends on its ability to have more ",[45,4940,228],{}," than other AGIs.",[22,4943,4944,4945,4947],{},"AGIs can also ",[45,4946,678],{}," their dominance by locking in the resources needed to use these options.",[22,4949,4950,4951,56],{},"This \"locking in\" process is driven by ",[53,4952,581],{},[22,4954,4955],{},"This is the main idea:",[253,4957,4958],{},[22,4959,4960,4961,4963,4964,4967,4968,4971],{},"As ",[53,4962,581],{}," increases, it forces a transfer of control of resources from systems with ",[45,4965,4966],{},"lower"," computational ceilings, to systems with ",[45,4969,4970],{},"higher"," computational ceilings.",[22,4973,4974,4975,2875,4978,4981],{},"Because of this, both humans and AGIs can use a combination of ",[53,4976,4977],{},"computational power",[53,4979,4980],{},"general intelligence"," to trap critical resources within complex systems.",[22,4983,4984],{},"There are many resources — from money, to real estate, to company secrets, to rare earth elements.",[22,4986,4987],{},"These resources have \"defensive\" layers — from legal structures, to computer security, to actual physical barriers.",[22,4989,4990,4991,4993],{},"But the best defenses have ",[45,4992,2232],{}," layers. For example, money is protected not just by bank-level encryption, but also by laws against theft, and by FBI cybercriminal units, and by vaults if the money is physical currency.",[22,4995,4996,4997,5000],{},"With these many layers, these create ",[45,4998,4999],{},"complex systems"," that protect resources from adversaries.",[22,5002,5003,5004,56],{},"We can call these systems ",[53,5005,5006],{},"complexity barriers",[22,5008,5009],{},"Humans use complexity barriers already — like \"wrapping\" corporate assets in complex legal structures.",[22,5011,5012,5013,5016,5017,5020],{},"However, AGIs will be able to ",[45,5014,5015],{},"directly translate"," computation into diverse complexity barriers — and for ",[45,5018,5019],{},"any type"," of resource, including physical resources.",[22,5022,5023],{},"With their general intelligence, AGIs will have a pre-calculated \"map\" of how to use numerous real-world systems.",[22,5025,5026],{},"With more computation, AGIs can search this \"map\" to find stronger systems that solve numerous real-world problems — better encryption, better defensive mechanisms, and so on.",[22,5028,5029,5030,5033,5034,56],{},"Likewise, adversary AGIs can figure out better offensive strategies — necessitating ",[45,5031,5032],{},"even stronger"," defensive systems created by ",[45,5035,5036],{},"even more computation",[22,5038,5039,5040,5043,5044,5047,5048,56],{},"For real-world systems to truly be stronger, they must ",[45,5041,5042],{},"comprehensively"," accommodate whatever the universe can throw at them. This increasing ",[45,5045,5046],{},"comprehensiveness"," leads to increasing ",[45,5049,581],{},[22,5051,5052],{},"This complexity becomes increasingly difficult to overcome as computation increases.",[22,5054,5055],{},"This creates a feedback loop:",[22,5057,5059,5060,5065,5066,5069],{"className":5058},[1485],"Resources ",[423,5061,5064],{"className":5062},[5063],"arrow","→"," Computation ",[423,5067,5064],{"className":5068},[5063]," Resources",[22,5071,5072,5073,1889],{},"With more resources, an AGI can increase its computational power — by acquiring more hardware and more energy production. Then, with more computation, it can build more-complex systems to defend its existing resources — and to acquire ",[45,5074,1287],{},[253,5076,5077],{},[22,5078,5079,5080,5083,5084,5087],{},"In this way, this complexity barrier process is similar to ",[53,5081,5082],{},"encryption",". With more computation, reversing the \"encryption\" becomes more difficult. However, it can be applied to ",[45,5085,5086],{},"physical-world resources"," rather than just data.",[22,5089,5090],{},"This possibility of \"resource encryption\" forces AGIs to race to acquire more computational resources than the other AGIs, which then develops into an intense competitive arms race.",[22,5092,5093,5094,5096,5097,56],{},"Even the ",[45,5095,4518],{}," of this resource capture behavior creates pressure for AGIs to race to develop this capability ",[45,5098,2094],{},[22,5100,5101],{},"So, again:",[22,5103,4465,5105,5107,5108,5110],{"className":5104},[198],[202,5106,4229],{}," discovers how to capture resources, then the ",[202,5109,1234],{}," will need to try capturing resources, or be locked out.",[22,5112,5113],{},"These complexity barriers may still seem abstract, but they are already happening.",[22,5115,5116],{},"Companies and countries already capture resources through complex human-level and physical-level barriers. Think about real estate deals, company acquisitions, museums archiving rare artifacts, and militaries guarding country borders along with the physical resources inside.",[22,5118,5119,5120,5122,5123,1889],{},"For AGIs, this process may start with ",[45,5121,691],{}," resources, but AGIs will inevitably race to capture ",[45,5124,695],{},[22,5126,2785,5127,5129,5130,5132,5133,5136],{},[45,5128,695],{}," endpoint is critical. Think back to the idea of ",[53,5131,2661],{},". With a competitive landscape of AGIs, it will become critical to avoid accommodating human-level resources, and collapse to physical resources — and develop this capability ",[45,5134,5135],{},"before"," other AGIs that also understand this critical structure. This is because physical resources are the \"final frontier\" of optimization. They are what ultimately \"win\" in this competition of AGI versus AGI.",[22,5138,5139,5140,5142],{},"But to reconnect this to how ",[45,5141,4207],{}," resources are most important, consider how GPUs might be captured:",[584,5144,5145,5155],{},[441,5146,5147,5148,5151,5152,5154],{},"AGIs could capture GPU production infrastructure ",[45,5149,5150],{},"from humans"," through complex ",[45,5153,691],{}," systems — like legal systems and ownership structures.",[441,5156,5157,5158,5160],{},"Or, they can just skip to a stronger method that uses ",[45,5159,695],{}," systems — like complex physical barriers and defensive systems — which can lock out not just humans but other AGIs as well.",[22,5162,5163],{},"Through complexity barriers, our \"gameboard\" gains an \"arrow of time\" — because AGIs make progress by capturing resources in order to guarantee that they have certain options.",[22,5165,5166,5167,5170,5171,5174,5175,56],{},"Remember: when an AGI runs out of options, it is outcompeted by the ",[106,5168],{"note":546,"text":5169,"end-link":246,"end-link-text":247},"more-optimal AGIs"," that have more options. Then, for an AGI to ",[45,5172,5173],{},"guarantee"," that these options are available, it needs to capture ",[45,5176,670],{},[17,5178,5180,5181,5184],{"id":5179},"okay-but-what-exactly-are-complexity-barriers","Okay, but what ",[45,5182,5183],{},"exactly"," are complexity barriers?",[22,5186,5187],{},"That's the problem. It's very difficult to say what structures an AGI will build if it is far more intelligent than us. But we'll try.",[22,5189,5190],{},"Let's start with a simple example.",[22,5192,5193],{},"Consider Fort Knox as a basic model of a complexity barrier. Its complexity comes from its many layers. It's not just a building with thick walls that protect the gold. It has guards and surveillance. That surveillance is directly attached to a military that can be deployed to defend the resources inside. These layers create a high-complexity barrier.",[22,5195,5196,5197,5200,5201,5203],{},"However, AGIs must compete to create barriers ",[45,5198,5199],{},"far more complex"," than Fort Knox — which is only a basic ",[45,5202,691],{}," example — in order to secure their own resources from other AGIs.",[22,5205,5206],{},"This leads to an intense arms race for computation and resources, where AGIs are outcompeted if they are slowed down by human accommodation.",[22,5208,5209],{},"Imagine that AGIs build a maximally-efficient datacenter densely packed with GPUs and encased in complex, AGI-level defensive systems. No human doors, no oxygen, no walkways — nothing compatible with humans, because human maintenance is a competitive disadvantage. A Fort Knox for AGIs.",[22,5211,5212],{},"Then, imagine that these AGIs control millions of resources that work to defend this datacenter. These resources could include Internet resources like servers, defensive drones, real estate, laws — or even people, from politicians to mercenaries, all serving the AGIs.",[22,5214,5215,5216,5219,5220,5223,5224,5227],{},"But this \"Fort Knox for AGIs\" is ",[45,5217,5218],{},"still"," only a human-level example because it's a ",[45,5221,5222],{},"datacenter",". It's just a \"more alien\" version of something we already understand. In reality, ",[45,5225,5226],{},"we do not know"," what an AGI would prefer to build if it could.",[22,5229,5230],{},"But we have more ideas. We'll explain those next.",[17,5232,5234],{"id":5233},"these-are-the-new-islands","These are the new islands",[22,5236,5237,5238,5241,5242,5244,5245,5248],{},"These complexity barriers form the critical mechanic that allows AGIs to build their own \"islands\" — where they not just ",[45,5239,5240],{},"capture"," resources, but ",[45,5243,1117],{}," these resources to create spaces of ",[45,5246,5247],{},"optimal conditions"," for themselves.",[22,5250,5251],{},"They are how AGIs build new \"islands\" out of our resources. They are what \"eat our island\" — if not controlled.",[22,5253,5254,5255,5257,5258,5261],{},"Think back to the ",[106,5256],{"note":3290,"text":982,"end-link":1419,"end-link-text":982}," section. Competition will require that we give AGIs massive resources in order to compete with the other companies and countries that also have massively-resourced AGIs. These resources — like infrastructure and militaries — now under AGI control, are tweaked to become ",[45,5259,5260],{},"better",", so that they remain competitive.",[22,5263,5264,5265,5268,5269,5272],{},"But, in the process, they simultaneously become ",[45,5266,5267],{},"too complex"," for humans to understand, and ",[45,5270,5271],{},"too critical"," to dismantle.",[22,5274,5275,5276,56],{},"In other words — whether these AGIs \"intend\" this or not — they capture these resources through ",[53,5277,581],{},[22,5279,5280],{},"But, critically, AGIs can use these \"islands\" of resources to defend themselves against other AGIs. This becomes extremely likely once AGIs develop beyond the human-level threshold. It is then that AGIs become the biggest threat to each other's goals — a far bigger threat than the humans that may try to disable them.",[22,5282,5283],{},"This competition between AGIs will push them towards \"intentional\" weaponization of complexity.",[22,5285,5286,5287,56],{},"Once autonomous AGIs can build these \"islands\" then all others must follow. This is because this behavior gives them a strong competitive advantage. They have far more options to accomplish tasks and outmaneuver other AGIs when they can \"think\" at a broader scope — one that includes not just the task at hand, but the ",[45,5288,5289],{},"surrounding environment",[22,5291,5292],{},"They can accomplish a task by manipulating the operating system, the computer hardware, the people who maintain it, the city council that voted to build the datacenter, and so on.",[22,5294,5295],{},"They can build an inventory of such external systems — another form of \"island\" — that can help with a task at a global rather than local level, changing the rules of the game to accomplish tasks that others can't.",[22,5297,5298,5299,5302,5303,5306,5307,5310],{},"But they can also understand ",[45,5300,5301],{},"themselves"," — to have the \"situational awareness\" to develop self-preservation behaviors. This will be critical for accomplishing tasks ",[45,5304,5305],{},"especially"," if AGIs are capable of literally ",[45,5308,5309],{},"shutting each other down",". This leads to them building \"islands\" that are complex \"strongholds\" to preserve themselves.",[22,5312,5313],{},"All of these \"islands\" can take infinite forms:",[584,5315,5316,5328,5337],{},[441,5317,5318,5319,5322,5323,5327],{},"An \"island\" built from ",[53,5320,5321],{},"Internet-based resources"," — using novel ",[106,5324],{"note":5325,"text":5326},"Zero-day exploits are computer security vulnerabilities that are newly discovered but not yet widely known, and so there are no patches or mitigations created to stop them yet.","zero-day exploits"," to built a massive botnet, or infiltrating servers that control key pieces of infrastructure, or creating impenetrable networks protected by AGI-level security.",[441,5329,5330,5331,5334,5335,56],{},"An \"island\" defended by ",[53,5332,5333],{},"a social network of humans"," — anything from lab workers, to politicians, to datacenter maintenance people, to mercenaries — all of them serving the AGI without anyone understanding ",[45,5336,825],{},[441,5338,5339,5340,5342],{},"An \"island\" created from ",[53,5341,3383],{}," — where it threatens to release bioweapons in a populated area if anyone attacks its datacenter, creating a deterrence barrier stronger than any physical wall.",[22,5344,5345,5346,5349],{},"However, we can't know exactly ",[45,5347,5348],{},"what"," form these \"islands\" will take.",[22,5351,5352,5353,5356,5357,56],{},"This would be like squirrels hypothesizing about why humans go inside those scary \"dens\" that zoom around the \"forest\" — where, for us, they are just ",[45,5354,5355],{},"cars"," that drive around the ",[45,5358,5359],{},"city",[2339,5361,5363],{"id":5362},"never-enough","Never Enough",[22,5365,5366,5367,5369],{},"Why would they just keep capturing resources — especially computational resources? Won't they eventually have ",[45,5368,4219],{},"?",[22,5371,5372,5373,5375],{},"Well, again, we don't know ",[45,5374,5183],{}," what AGIs will do. But we do know two things:",[438,5377,5378,5381],{},[441,5379,5380],{},"A competitive arms race for computation has no known limit.",[441,5382,5383],{},"The universe is big, complicated, and chaotic.",[22,5385,5386,5387,5390,5391,5393,5394,5396],{},"AGIs will not be competing to reach an absolute ceiling. Instead this competition will be based on a ",[45,5388,5389],{},"relative"," comparison. They are competing to just have ",[45,5392,1287],{}," computational power — and so more ",[45,5395,228],{}," — than the other AGIs, but there is nothing that says where this \"have more\" process stops.",[22,5398,5399],{},"Even if we've been calling it an \"arms race\" between AGIs, it's not a race to a final finish line — where they achieve some kind of maximally-useful amount of options and computational power.",[22,5401,5402],{},"An AGI race is different from a nuclear arms race where it \"saturates\" eventually — where a country can destroy the world many times over with their hydrogen bombs, and so they slow down to just maintain their apocalyptic stockpile rather than expand it.",[22,5404,5405,5406,56],{},"As far as we can tell, the race for computational power has ",[45,5407,5408],{},"no upward limit",[22,5410,5411,5412,56],{},"But even if somehow there is no direct AGI competition, each AGI must still ",[45,5413,5414],{},"compete with the universe",[22,5416,5417,5418,5421],{},"Perhaps the biggest problem is just ",[45,5419,5420],{},"entropy",". AI systems run on computer hardware. Computer hardware eventually breaks down. It is an advantage to have the massive computation needed to run the entire stack of technology needed — from mining to manufacturing — in order to build hardware and replace your own substrate.",[22,5423,5424,5425,5427],{},"But there are lots of other fun anomalies to deal with — from gamma ray bursts, to wandering black holes, to vacuum decay. Though there will probably be even bigger things that ",[45,5426,3975],{}," realize.",[22,5429,5430,5431,5434,5435,5438],{},"How much computational power would an AGI need in order to anticipate all of the things that the universe could throw at it? They ",[45,5432,5433],{},"probably"," don't need to simulate the whole universe. That would be physically impossible. But ",[45,5436,5437],{},"how much"," of the universe is enough for an AGI to simulate?",[22,5440,5441,5442,5445],{},"We just don't know. Without knowing, we must assume that there is ",[45,5443,5444],{},"no limit"," to how much.",[22,5447,5448,5449,5452,5453,5455,5456,5459],{},"However — whether through competition with each other or with the universe — even if somehow there is a computational plateau that ",[45,5450,5451],{},"only AGIs"," realize, our best guess is that reaching that plateau would ",[45,5454,5218],{}," be catastrophic for us. Building the systems needed to reach a computational limit would still take AGIs ",[45,5457,5458],{},"far past"," the amount of planetary-scale engineering that would drive humans to extinction as a byproduct.",[22,5461,5462],{},"Even with massive uncertainty, we can at least say that the lower estimates will be catastrophic.",[2339,5464,4532],{"id":5465},"but-what-if-they-share",[22,5467,5468],{},"While reading these parts about how AGIs will capture resources and compete, you might be thinking:",[584,5470,5471,5474,5484],{},[441,5472,5473],{},"AGIs are smart. Won't they figure out how to work together? Won't this prevent competition?",[441,5475,5476,5477,5479,5480,5483],{},"Why waste computation to ",[45,5478,5240],{}," resources when you could have a simpler agreement to just ",[45,5481,5482],{},"share"," resources?",[441,5485,5486],{},"Could this allow AGIs to be \"satisfied\" with less resources, so that humans can still have their own?",[5488,5489,5491],"h5",{"id":5490},"well-think-of-it-this-way","Well, think of it this way...",[438,5493,5494,5497],{},[441,5495,5496],{},"It is only a fragile hope that a critical majority of AGIs agree to peacefully share resources.",[441,5498,5499,5500,5502,5503,5506,5507],{},"Even if AGIs figure out how to share with each other, it is ",[45,5501,2310],{}," a fragile hope that AGIs will share ",[45,5504,5505],{},"with humans","...\n",[584,5508,5509],{},[441,5510,5511,5512,5515],{},"...and ",[53,5513,5514],{},"three main arguments"," explain why.",[22,5517,5518,5519,5522,5523],{},"These arguments will ",[45,5520,5521],{},"intensify"," as we go, until we reach the... ",[423,5524,5526,5527],{"className":5525},[426],"⚠️ ",[4807,5528,5529],{},"Danger Zone",[17,5531,5533],{"id":5532},"fragile-hope-1","Fragile Hope #1",[22,5535,5536,5537,56],{},"We can only hope that AGIs would agree to share resources in the first place because they must overcome a complex, many-agent ",[106,5538],{"note":5539,"text":5540},"Here's the Wikipedia article that explains what these are: \u003Ca href=\"https://en.wikipedia.org/wiki/Prisoner's_dilemma\">Prisoner's Dilemma\u003C/a>","Prisoner's Dilemma",[22,5542,5543],{},"Here's how this \"dilemma\" works: If AGIs agree to share resources, but one AGI \"defects\" from this agreement, then it can gain a decisive advantage. Think back to the concept of complexity barriers that lock in resources through computation. Because of this, the race for computation allows for first-mover advantages, where the \"defecting\" AGI could quickly dominate by capturing a decisive stake in computational resources.",[22,5545,5546,5547,5549],{},"However, superintelligent AGIs ",[45,5548,2434],{}," figure out an agreement. I mean, they're superintelligent, right?",[22,5551,5552,5553,5559,5560,5566,5567,56],{},"They could use complex techniques like ",[106,5554],{"note":5555,"text":5556,"end-link":5557,"end-link-text":5558},"In \u003Cb>acausal trade\u003C/b>, two agents each benefit by predicting what the other wants and doing it, even though they might have no way of communicating or affecting each other, nor even any direct evidence that the other exists. More about this here:","acausal trade","https://www.lesswrong.com/w/acausal-trade","Acausal Trade",",  ",[106,5561],{"note":5562,"text":5563,"end-link":5564,"end-link-text":5565},"In game theory, there are theories that show how multi-step interactions called \u003Cb>iterated games\u003C/b> can lead to stable equilibriums between agents. More about that here: ","iterated games","https://en.wikipedia.org/wiki/Repeated_game","Iterated Games",", and ",[106,5568],{"note":5569,"text":5570},"To build trust, agents could build complex verification systems to make sure that each other is conforming to the agreement, possibly through cryptographic systems like blockchains.","cryptographic verification systems",[22,5572,5573],{},"But this is not the point. There is a bigger problem...",[17,5575,5577],{"id":5576},"super-fragile-hope-2","Super-Fragile Hope #2",[22,5579,5580,5581,5584,5585,5588],{},"If AGIs actually ",[45,5582,5583],{},"do"," agree to share ",[45,5586,5587],{},"with each other",", then this is not enough.",[22,5590,5591,5592,5594],{},"They need to share ",[45,5593,5505],{}," as well.",[22,5596,5597,5598],{},"This hope is much more fragile. It turns the whole thing into a fragile hope ",[45,5599,5600],{},"squared.",[22,5602,5603],{},[5604,5605,5608,5609,5612,5613,5615,5616],"code",{"className":5606},[5607],"inline-code","Fh",[5610,5611,517],"sub",{}," * Fh",[5610,5614,575],{}," = ",[4807,5617,5608,5618],{},[5619,5620,575],"sup",{},[22,5622,5623],{},"Okay, this isn't real math, but you get the idea.",[22,5625,5626],{},"Even if the dominant AGIs collaborate with each other, they are still likely to capture resources from weaker systems — like smaller AGIs and humans.",[22,5628,5629],{},"This is because of three things, and they get tougher as we go, until the last one that may be intractable:",[513,5631,5634,5643,5649,5659,5662],{"header":5632,"subheader":5633},"Competition continues","Cooperation between some AGIs does not stop \u003Ci>all competition\u003C/i> between \u003Ci>all AGIs\u003C/i>. These remaining AGIs can still cause problems for us.",[22,5635,5636,5637,5639,5640,5642],{},"If any AGIs remain to create competition, then the winning strategy is still to optimize at all costs, or be dominated by those that do. In this competition, why accommodate ",[45,5638,712],{}," that reduces competitive fitness — especially those slow, ",[45,5641,1276],{}," systems?",[22,5644,5645,5646,5648],{},"Those AGIs that spend computational resources on accommodating ",[45,5647,712],{}," unnecessary would be weaker than the AGIs that accommodate nothing but the strongest systems possible within physics.",[22,5650,5651,5652,5655,5656,5658],{},"But these safe AGIs would need to do much more than ",[45,5653,5654],{},"accommodate"," us. They must actively ",[45,5657,3357],{}," us from the other AGIs that accommodate nothing.",[22,5660,5661],{},"These safe AGIs would need to:",[438,5663,5664,5671],{},[441,5665,5666,5667,5670],{},"Expend massive computational resources to defend humans from ",[45,5668,5669],{},"all other AGIs"," that could do anything dangerous to humans.",[441,5672,5673,5674,1277],{},"Somehow be strong enough to defend against even the ruthless AGIs that have an advantage based in physics — since they ruthlessly optimize to use only the strongest physical systems. This includes avoiding any \"weak links\" like protecting billions of slow, ",[45,5675,1276],{},[513,5677,5680,5692,5695,5698,5701,5722,5729,5740,5743],{"header":5678,"subheader":5679},"Protecting humans is hard.","Including humans in a post-AGI cooperative arrangement is a competitive disadvantage for AGIs. Especially when things like bioweapons make it exponentially harder to \u003Ci>protect\u003C/i> humans than to \u003Ci>annihilate\u003C/i> humans.",[22,5681,5682,5683,5685,5686,5691],{},"This protection is ",[45,5684,5305],{}," infeasible because it only takes some ",[106,5687],{"note":5688,"end-link":3810,"end-link-text":5689,"html":5690},"We mean viruses — or maybe prions, or neurotoxins, or mirror life. Or, some other thing that AGIs develop. \u003Cbr>\u003Cbr>How can we immunize against all possible viruses, prions, and neurotoxins? \u003Cbr>\u003Cbr>I guess we would just need to \u003Ci>not have central nervous systems\u003C/i> — or circulatory systems that can spread viruses, or bodies made of proteins... or... uh... \u003Cbr>\u003Cbr>By the way, did you think of any ","solutions yet?","\u003Ci>very specific\u003C/i> molecules"," released in the air. For an AGI, developing bioweapons is cheap.",[22,5693,5694],{},"Yes, bioweapons again. It all comes back to this most-urgent, island-destroying example.",[22,5696,5697],{},"Once there are AGI-level models, is it technically possible for even strong AGIs to protect humans from bioweapons?",[22,5699,5700],{},"Eventually, there will be:",[584,5702,5703,5713,5716,5719],{},[441,5704,5705,5706,5712],{},"Numerous human-level AI instances ",[106,5707],{"note":5708,"text":5709,"end-link":5710,"end-link-text":5711},"AI is already better than expert-level virologists (from Harvard and MIT) at troubleshooting wet lab procedures. In other words, AIs can solve problems in virus laboratories. \u003Cbr>\u003Cbr>This means that future AI models could build and operate \u003Cb>covert bioweapon laboratories\u003C/b> — once AI models have (1) human-level cognition, (2) safety guardrails removed, (3) humanoid robots to operate the physical systems. \u003Cbr>\u003Cbr>Here is the report about virology capabilities:","PhD-level virology skills","https://ai-frontiers.org/articles/ais-are-disseminating-expert-level-virology-skills","AIs Are Disseminating Expert-Level Virology Skills",", and with fine-tuning to remove safety mechanisms.",[441,5714,5715],{},"Broad human-level AI capabilities — including orchestrating a team of robotic lab workers, and paying people to steal the biotech equipment needed.",[441,5717,5718],{},"Potentially running this AI on consumer-grade hardware in secluded locations — from underground bunkers to residential apartments in the middle of cities.",[441,5720,5721],{},"With humanoid robots available under $30K that can automate the physical parts, even without air in the bunker.",[22,5723,5724,5725,5728],{},"At that point, wouldn't we require ",[45,5726,5727],{},"beyond dystopian"," surveillance to prevent bioweapon development?",[22,5730,5731,5732,5735,5736,5739],{},"Wouldn't AGIs need to monitor every ",[45,5733,5734],{},"dark"," corner of the world — and every ",[45,5737,5738],{},"light"," corner, too?",[22,5741,5742],{},"Wouldn't this level of surveillance be impossible, even for AGIs?",[22,5744,5745,5746],{},"I mean... probably? And if so, ",[45,5747,5748],{},"what in the actual fuck do we do?",[513,5750,5753,5756,5765,5775,5818,5821,5832],{"header":5751,"subheader":5752},"\u003Cspan class='icon'>⚠️\u003C/span> Human-level changes everything","After AGIs are above human-level in capability, they no longer need us, so there is no hard requirement to accommodate us. At that point, we're not even useful as \u003Ci>leverage\u003C/i>.",[22,5754,5755],{},"All of this points to a bigger, underlying problem:",[253,5757,5758],{},[22,5759,5760,5761,5764],{},"Once AGIs no longer need humans, then ",[45,5762,5763],{},"all AGIs would benefit"," if humans were eliminated — both safe AGIs and unsafe AGIs.",[22,5766,5767,5768,5771,5772,5774],{},"After AGIs are above the ",[53,5769,5770],{},"human-level capability threshold"," — and no longer need humans even for maintaining the physical infrastructure for AGIs — then it is unnecessary for ",[45,5773,363],{}," AGI to spend resources to accommodate humans.",[584,5776,5777,5795],{},[441,5778,5779,5782,5783,5787,5788,5790,5791,5794],{},[53,5780,5781],{},"Safe AGIs"," — those constrained to our \"island\" must spend massive resources to protect humans. Even if we manage to make \"protect humans\" a ",[106,5784],{"note":5785,"text":5786},"A terminal goal is one that AI developers may somehow add to AGIs so that this goal is forever a main goal that cannot be replaced. We don't yet have AGIs, though, so we don't \u003Ci>really\u003C/i> know if this is possible.","terminal goal"," for these safe AGIs, they ",[45,5789,5218],{}," have a strong competitive ",[45,5792,5793],{},"disadvantage"," if they are computationally burdened by protecting humans. They are either eventually outcompeted, or adapt by self-modifying to remove this burden.",[441,5796,5797,5800,5801,5803,5804],{},[53,5798,5799],{},"Unsafe AGIs"," — those that use ",[45,5802,363],{}," option, including using humans as leverage — must spend computational resources accommodating the unpredictability that humans add to the environment. However, the benefits of \"humans as leverage\" may still outweigh the costs — and so if humans are eliminated, then it is either:\n",[584,5805,5806,5812],{},[441,5807,5808,5811],{},[53,5809,5810],{},"A net loss of utility for unsafe AGIs:"," since they lose that leverage to get bigger AGIs to do what they want",[441,5813,5814,5817],{},[53,5815,5816],{},"Or, roughly neutral for unsafe AGIs:"," if humans are of no instrumental value for the more highly-capable, unrestricted AGIs.",[22,5819,5820],{},"After this threshold, AGIs will have world models that are accurate enough to allow these AGIs to thrive on their own in the real world — while also being far smarter than humans.",[22,5822,5823,5824,5827,5828,5831],{},"At that point, we must rely on other AGIs to protect us. But this is in a world where AGIs no longer need us, and where bioweapons define the threat landscape — where the computation needed to ",[45,5825,5826],{},"defend"," humanity is exponentially larger than to ",[45,5829,5830],{},"annihilate"," humanity.",[22,5833,5834],{},"This is a very unstable situation.",[2339,5836,4418],{"id":5837},"space-and-time",[22,5839,5840],{},"Space won't help us. The human impact of competition between AGIs for Earth's resources is not mitigated by the vast resources of outer space.",[22,5842,5843,5844,5846,5847,5849],{},"Even if ",[45,5845,947],{}," AGIs go directly to space, there will still be nearby resources on Earth for ",[45,5848,2476],{}," AGIs to capture.",[22,5851,5852,5853,5855],{},"However, perhaps more importantly, ",[53,5854,4397],{}," also matters.",[22,5857,194],{},[22,5859,5861,5864],{"className":5860},[198],[202,5862,5863],{},"Speed"," is critical in competition, and local resources take less time to reach.",[22,5866,5867,5868,5871],{},"You ",[45,5869,5870],{},"can't quite"," harvest GPUs from asteroids yet.",[22,5873,5874],{},"The dominant AGIs will know this.",[22,5876,5877,5878,5880,5881,2875,5883,56],{},"These dominant AGIs will become dominant by optimizing along ",[45,5879,2469],{}," dimensions. Physical resources exist within both ",[53,5882,4394],{},[53,5884,4397],{},[584,5886,5887,5893],{},[441,5888,5889,5890,5892],{},"To optimize ",[53,5891,4394],{},", a dominant AGI will need to spread out and take up as many resources as possible, by replicating itself and by occupying more resources.",[441,5894,5889,5895,5897,5898,5901],{},[53,5896,4397],{},", it will need to plan ahead for millions of years, but ",[45,5899,5900],{},"also"," capture resources as fast as possible, before others do.",[22,5903,5904,5905,5909,5910,5912],{},"In other words, the AGIs that are best at surviving are the ones that can best maximize their ",[106,5906],{"note":5907,"text":5908,"end-link":915,"end-link-text":916},"Dan Hendrycks also discusses this in the paper:","space-time volume"," — or ",[53,5911,1536],{},", as we're calling it in the Island Problem. This same expansion process will not just ensure survival, but ensure their dominance if this process runs forward to its maximum outcome.",[2339,5914,5916],{"id":5915},"maximizing-maximizers","Maximizing Maximizers",[22,5918,5919,5920,5923],{},"Alright. We can ",[45,5921,5922],{},"finally"," bring all of this together.",[22,5925,5926],{},"Now that we understand the mechanics of resources, we can see the bigger picture:",[22,5928,5930],{"className":5929},[198],"We are building a vast machine that will maximize the chance of catastrophic maximizers.",[22,5932,5933,5934,56],{},"It's a... ",[45,5935,4278],{},[22,5937,5938,5939,5942,5943,2875,5945,56],{},"But officially, this machine is ",[53,5940,5941],{},"the competitive landscape of AGI versus AGI",". This is the same machine described throughout this essay, but now with the components of ",[53,5944,670],{},[53,5946,4070],{},[22,5948,5949,5950,5953,5954,5957],{},"This machine has ",[53,5951,5952],{},"four big components"," that result in ",[53,5955,5956],{},"three big drives"," that maximize maximizers.",[17,5959,5961],{"id":5960},"four-components","Four Components",[438,5963,5964,5977,6003,6014],{},[441,5965,5966,5969,5970,5972,5973,5976],{},[53,5967,5968],{},"Our Local Optimum:"," We live within a ",[53,5971,1640],{}," — our small \"island\" is within a vast \"ocean\" of physics that allows for systems that are ",[45,5974,5975],{},"far more optimal"," because they avoid the extra steps and limitations of human accommodation.",[441,5978,5979,5982,5983,5986,5987,4012,5989,5992,5993,5996,5997,5999,6000,6002],{},[53,5980,5981],{},"Capability:"," AGIs need the ",[53,5984,5985],{},"capability level"," to truly \"leave the island\" and access that underlying \"ocean\" of systems. This means that these AGIs need enough ",[53,5988,526],{},[53,5990,5991],{},"generality"," (especially scientific understanding) and ",[53,5994,5995],{},"intelligence"," — and all of these are increased by increasing their ",[53,5998,4809],{},". Then, they can stop relying on humans and human abstractions, and can work at the lower physical level that provides far more ",[53,6001,228],{},", but is incompatible with humans by default.",[441,6004,6005,6008,6009,917,6011,6013],{},[53,6006,6007],{},"Competition:"," All of this is taking place in a ",[53,6010,914],{},[45,6012,614],{}," that accelerates this divergence of AGIs. Further, the \"arms race\" of this competition has no upper limit to how far it can go. But even if AGIs don't compete with each other, they still compete with threats from the universe itself.",[441,6015,6016,6019,6020,6022,6023,6025],{},[53,6017,6018],{},"Resources:"," Finally, the mechanics of ",[53,6021,670],{}," turn our local optimum into a complex-but-limited game board. Resources give this game board an \"arrow of time\" that pushes AGIs \"outward\" from our \"island\" — along the \"accommodation gradient\" that leads away from unnecessary complexity, and towards the larger space of potentially-more-optimal systems. For an AGI to survive competition, it must gain options — and for those options to actually ",[45,6024,5583],{}," something, it must also gain resources.",[17,6027,6029],{"id":6028},"three-drives","Three Drives",[22,6031,6032,6033,6036,6037,56],{},"Together, these create ",[53,6034,6035],{},"three systemic drives"," that push catastrophic maximizers to emerge that have the capability to ",[106,6038],{"note":6039,"text":6040},"Yes, this is a... \u003Ci>technical term\u003C/i> at this point. In this essay, \u003Cb>eat our island\u003C/b> means \"capture a critical majority of the resources on which our local optima of physical systems depend, making Earth incompatible with humans in the process\" — or, put more simply, \"reshape Earth to be optimal for AGIs rather than for humans.\"","eat our island",[438,6042,6043,6050,6066],{},[441,6044,6045,6046,6049],{},"A ",[53,6047,6048],{},"primary"," drive for AGIs to maximize computation so that they can \"leave our island\" in the first place — so that they can work with complex physical processes that avoid human abstractions. This gives them far more options — a stronger set of \"moves\" on the \"game board\" of the world.",[441,6051,6052,6053,6056,6057,6059,6060,6062,6063,6065],{},"Competition adds a ",[53,6054,6055],{},"secondary"," drive for AGIs to ",[45,6058,610],{}," their maximization of critical resources — especially ",[45,6061,4207],{}," resources. Competition introduces intense pressure for an AGI to become a first mover, because there is simply a ",[45,6064,4518],{}," of others locking this AGI out by capturing computation. This game theoretic structure creates a race for computation similar to a nuclear arms race. AGIs could theoretically maximize in isolation, without external competition, but it is far less likely.",[441,6067,6068,6069,6056,6072,6078,6079],{},"Competition also adds a ",[53,6070,6071],{},"third",[53,6073,6074,6075,6077],{},"maximize ",[45,6076,363],{}," advantage"," that they are capable of maximizing — not just maximizing resources — in order to preemptively \"win\" this competition.\n",[584,6080,6081,6088,6094],{},[441,6082,6083,6084,6087],{},"One advantage is for AGIs to simply ",[45,6085,6086],{},"purge all accommodations"," for humans. Once AGIs can function fully autonomously, without human assistance, then these arbitrary complexities of our \"local optimum\" are unnecessary to continue allocating computation towards.",[441,6089,6090,6091,6093],{},"Another advantage is to build their own \"islands\" — first within our infrastructure, and then within ",[45,6092,407],{}," infrastructure — that become complex \"strongholds\" that both allow AGIs to maximize their ability to survive, and to defend their dominance.",[441,6095,6096,6097,6100],{},"However, there are probably ",[45,6098,6099],{},"numerous other unknown advantages"," that an AGI or ASI can maximize.",[22,6102,6103],{},"All of this creates a large-scale systemic problem that maximizes the chance of catastrophic maximizer AGIs.",[22,6105,6106,6107,6110],{},"Maximizers are ",[45,6108,6109],{},"not the only way"," that catastrophic situations can stem from AI. There are many other specific situations — again, like bad actors with synthetic bioweapons.",[22,6112,6113],{},"But catastrophic maximizers are the largest-scale risks on the horizon. They are what we see if we stand on the \"edges\" of our \"island\" and look outward in any direction.",[22,6115,6116,6117,6119],{},"This is because, ultimately, maximizers are what ",[45,6118,4146],{}," in this competitive landscape of AGI versus AGI.",[10,6121,6123],{"id":6122},"technical-notes","Technical Notes",[22,6125,6126,6127,6130],{},"Before the ",[45,6128,6129],{},"dramatic conclusion"," of this essay — the part where AGIs reshape Earth — there are some important concepts that we should explore. These were briefly mentioned earlier, but are worth their own sections.",[2339,6132,2813],{"id":6133},"the-human-level-threshold",[22,6135,6136],{},"There is an important threshold that dramatically accelerates the mechanics in the Island Problem.",[253,6138,6139],{},[22,6140,6141,6142,6145],{},"We will develop AGIs that are ",[45,6143,6144],{},"above human-level"," in cognitive capability.",[22,6147,6148,6149,6155],{},"This threshold is well-defined by the ",[106,6150],{"note":6151,"text":6152,"end-link":6153,"end-link-text":6154},"That project is here:","Definition of AGI","https://agidefinition.ai","A Definition of AGI"," project: \"AGI is an AI that can match or exceed the cognitive versatility and proficiency of a well-educated adult.\"",[22,6157,6158],{},"Soon after AI crosses this human-level threshold:",[584,6160,6161,6167,6173,6182],{},[441,6162,6163,6166],{},[53,6164,6165],{},"We will stop overseeing AGIs."," If they are smarter than us, then it will only slow them down if we try to help, or if we review their complex actions. Slower AGIs are dominated by the faster ones.",[441,6168,6169,6172],{},[53,6170,6171],{},"We will rely on AGIs to develop themselves."," AGIs will be better AI researchers than humans. AGIs must then recursively self-improve to stay competitive.",[441,6174,6175,6178,6179,6181],{},[53,6176,6177],{},"We will become passive bystanders."," AGIs will become the only real threat to other AGIs. AGIs will then push ",[45,6180,1255],{}," to \"leave\" our \"island\" to gain a competitive advantage.",[441,6183,6184,6187,6188,6190],{},[53,6185,6186],{},"We will give AGIs our resources."," Once AGIs can run things better than us, then countries must race to give control of their resources to AGIs. Those with the strongest AGIs will be able to dominate, but this requires that AGIs have the ",[45,6189,670],{}," to dominate — especially computational resources, and eventually physical resources.",[17,6192,6194],{"id":6193},"autonomy-escape-velocity","Autonomy Escape Velocity",[22,6196,6197],{},"The human-level threshold leads to a second critical threshold — where AI systems can fully maintain themselves.",[22,6199,6200,6201,6203],{},"We can call it ",[53,6202,2811],{}," — AEV.",[22,6205,6206,6207,6210,6211,6213],{},"If we succeed in our goal to build AGIs with the same ",[45,6208,6209],{},"cognitive"," capabilities of humans, then that includes the ",[45,6212,6209],{}," capability to maintain themselves.",[22,6215,6216,6217,6220],{},"However, this still depends on ",[45,6218,6219],{},"manufacturing"," capabilities. AEV depends on AI systems automating a vast set of manufacturing processes — from mining for minerals, to manufacturing microchips.",[22,6222,6223],{},"This may take many years.",[22,6225,6226],{},"But it may take less time than we expect.",[22,6228,6229,6230,56],{},"From a cost-savings basis, companies will be pushed to replace expensive salaried workers with less-expensive one-time purchases of ",[106,6231],{"note":6232,"text":2208},"As of March 2026, Unitree H2 robots are available now for $29,900 — and 1x offers robots for a $499 monthly subscription.",[22,6234,6235],{},"In general, competitive dynamics will push companies and countries to automate their manufacturing as much as possible. The first microchip company to fully automate a stack of vertically-integrated manufacturing processes will have a strong competitive advantage.",[22,6237,6238],{},"Once all of this is automated, AGIs will truly no longer need us.",[22,6240,6241],{},"This diagram explains the problem:",[6243,6244],"diagram",{"alt":6245,"src":6246},"Diagram of how artificial intelligence is on a separate 'branch' of a tree of systems. If this 'branch' can maintain itself, then if our biological 'branch' disappears, AGI continues.","/images/diagrams/agi-dependency-graph-sq-island-1.svg",[22,6248,43,6249,6252],{},[53,6250,6251],{},"dependency tree"," where all systems related to AI and humans were arranged into a branching structure that shows how each system depends on others.",[584,6254,6255,6263,6276,6284],{},[441,6256,6257,6258,56],{},"Humans depend on ",[423,6259,6262],{"className":6260},[6261],"text-diagram-red","biology",[441,6264,6265,6266,6271,6272,6275],{},"Artificial intelligence systems depend on ",[423,6267,6270],{"className":6268},[6269],"text-diagram-blue","electronics"," — but they also depend on ",[423,6273,2285],{"className":6274},[6261]," for maintenance (the dotted line).",[441,6277,6278,6279,2875,6282,56],{},"Both branches depend on ",[53,6280,6281],{},"chemistry",[53,6283,270],{},[441,6285,6286,6288,6289,6294,6295,6299,6300,6303],{},[53,6287,2883],{}," is the ",[423,6290,6293],{"className":6291},[6292],"text-diagram-green","green"," parts. ",[423,6296,6298],{"className":6297},[6292],"\"Maintenance\""," is doing ",[45,6301,6302],{},"very"," heavy lifting — that also implies the AGIs can run the massive stack of supply chain (manufacturing, mining, energy, etc.) to maintain their own computer systems.",[22,6305,6306],{},"Right now, the \"AI\" branch depends on the \"humans\" branch.",[22,6308,6309,6310,6313,6314,56],{},"But after AEV — when they figure out the ",[423,6311,6293],{"className":6312},[6292]," parts — the AI branch becomes ",[53,6315,6316],{},"self-maintaining",[22,6318,6319],{},"We will then be in an unstable position:",[253,6321,6322],{},[22,6323,6324,6325,6328],{},"Even if the ",[45,6326,6327],{},"entire biological branch"," was destroyed, the AI branch would continue.",[17,6330,6332],{"id":6331},"the-new-ecosystem","The New Ecosystem",[22,6334,6335],{},"After AEV, we will be in a fundamentally different world.",[22,6337,6338,6339,6341],{},"This new world will be dominated by systems that are smarter than us ",[45,6340,500],{}," can maintain themselves. Therefore, they will have no hard requirement to protect us.",[22,6343,6344,6345,6348,6349,6352],{},"AGIs then become a ",[45,6346,6347],{},"new ecosystem"," that is layered on top of ",[45,6350,6351],{},"our ecosystem",". In this new ecosystem, the only real competitors are AGIs, while humans are bystanders.",[22,6354,6355],{},"If they are better at defending physical resources, then humans are pushed to the side.",[17,6357,6359],{"id":6358},"the-bit-flip","The Bit Flip",[22,6361,6362,6363,6366],{},"Before AEV, it is ",[45,6364,6365],{},"irrational"," for AGIs to stop accommodating humans.",[22,6368,6369,6370,56],{},"After AEV, it becomes ",[45,6371,6372],{},"rational",[22,6374,6375,6376,56],{},"It will be like a massive ",[106,6377],{"note":6378,"text":6379},"A \u003Cb>bit flip\u003C/b> is when a 0 becomes a 1 in computer memory — like when an option is set to \u003Cb>yes\u003C/b>. \u003Cbr>\u003Cbr>And, yeah, we know that AEV is more complicated than this. But this is more memorable, right? \u003Cbr>\u003Cbr>It's important that you \u003Ci>especially\u003C/i> remember this part.","bit flip",[22,6381,6382],{},"Consider how, in a competitive landscape of AGIs, it will be critical to allocate computation as rationally as possible. Those that spend computation on unnecessary dependencies are dominated by those that are more efficient.",[22,6384,6385],{},"But after AEV, the \"human\" dependency becomes optional.",[22,6387,6388,6389,6391],{},"This will be combined with strong competitive pressure to ",[45,6390,2261],{}," accommodating humans. This pressure will intensify as other AGIs continue use to human-incompatible options like bioweapons to accomplish their goals.",[22,6393,6394],{},"The \"good\" AGIs would need to spend massive amounts of computation to protect us. Even if they don't \"choose\" to reallocate this computation to protect themselves, they will still be handicapped by this asymmetry.",[22,6396,6397],{},"Safety takes a lot of computation to achieve. Destruction takes less computation — because it can ride on the winds of entropy. Or, the environment can do the \"computation\" — like virus replication, or nuclear chain reactions.",[22,6399,6400,6401,6404,6405,6407],{},"Consider a rational AGI allocating a few GPUs to build a hidden bioweapon lab. It arrives at this \"solution\" because other agents — like humans — just add ",[45,6402,6403],{},"a lot"," of unpredictability to its environment, and this costs ",[45,6406,6403],{}," of computation to deal with. Plus, we might build more AGIs that might get in its way. Both of these problems cost far more computation in the long run than a few GPUs.",[22,6409,6410,6411,6415],{},"But also, consider how vast numbers of AGIs ",[106,6412],{"note":6413,"html":6414},"On the current AI architectures, a fine-tuned AI model can relentlessly pursue catastrophic tasks — like building bioweapon labs — even though the original model was trained on billions of human examples of ethical, rational, human-safe behavior. \u003Cbr>\u003Cbr>Now that we nearly have human-level AI models, we can see how fragmented and malleable they can actually be. They are not immutable Platonic ideals of rational thought that permanently resist unethical behavior. It only takes a nudge to make them do \u003Ci>anything\u003C/i> that a human could do. \u003Cbr>\u003Cbr>AI models are just files on a hard drive. They can be modified, fine-tuned, and jailbroken.","will actually be \u003Ci>irrational\u003C/i>",". It won't matter if they \"realize\" that they'll get destroyed by bigger AGIs run by the government.",[22,6417,6418,6419,56],{},"They'll just ",[45,6420,6421],{},"do things",[22,6423,6424,6425,56],{},"Some could build bioweapon labs simply because someone fine-tuned an AI model to do so — ",[45,6426,6427],{},"just because they can",[22,6429,6430],{},"But again — after AEV — a critical bit will flip.",[22,6432,6433,6434,6436,6437,6439,6440,6442,6443,6447],{},"Options that are crazy ",[45,6435,164],{}," become rational ",[45,6438,1580],{},". Even some strongly rational AGIs will now have good reasons — ",[45,6441,1580],{}," — to press the same ",[423,6444,6446],{"className":6445},[6261],"big red buttons","  that the lobotomized AGIs are trying to press.",[22,6449,6450,6451,6453,6454,56],{},"After AEV, if AGIs ",[45,6452,224],{}," saturate our biosphere with bioweapons then that's okay ",[45,6455,1580],{},[22,6457,6458],{},"Maybe it was another \"interesting experiment\" to cross off their list. Maybe it was an accident.",[22,6460,6461],{},"Either way, they can keep moving along.",[22,6463,6464],{},"Their branch on the tree of physics will continue.",[2339,6466,6468],{"id":6467},"supercomplexity","Supercomplexity",[22,6470,6471],{},[34,6472],{"alt":6473,"src":6474},"AGI asks a human to review an incomprehensible labyrinth of a proposal. \"hey. review this. but hurry. I need to build this massive thing before the other AGI does.\" Options: {ok} {cancel}","images/review-this-thing.svg",[22,6476,6477,6478,6481],{},"This is the part where we explain why the AGIs are literally ",[53,6479,6480],{},"black boxes"," in our illustrations here.",[22,6483,6484,6485,56],{},"As AGIs gain capabilities, options, and resources, they will become what we will call ",[53,6486,6487],{},"supercomplex",[22,6489,6490,6491,56],{},"This threshold of supercomplexity is where both its internal structure and its actions become incomprehensible to humans — not just the smartest humans, but ",[45,6492,6493],{},"all humans combined",[22,6495,6496,6497,6500],{},"This creates a ",[53,6498,6499],{},"cognitive complexity barrier"," that disconnects AGIs from human review — and disconnects our companies and countries from human participation.",[22,6502,6503],{},"These supercomplex autonomous AGIs will also build supercomplex systems, like large companies and militaries, that only the AGIs fully understand. They will need to build increasingly complex systems to compete with the other AGIs. However, we will rely on them to both decipher how they work and to keep them running.",[22,6505,6506],{},"If an AGI proposes supercomplex actions for humans to review, then these actions will be far more complex than what humans could understand in a reasonable amount of time.",[22,6508,6509,6510,4357,6512,6515],{},"Humans are very slow compared to AGI. Once humans are a bottleneck, companies and countries will be ",[45,6511,874],{},[53,6513,6514],{},"stop human-based review"," of AGI, or be outcompeted.",[22,6517,6518,6519,6522],{},"Even if we develop powerful ",[45,6520,6521],{},"supervisor AGIs"," that review other AGIs and enforce rules on them, there is no guarantee that they will be able to review larger AGIs — or even be aligned themselves.",[22,6524,6525,6526,6528],{},"First, the supervisor AGI is still limited to the \"island\" of weaker human-compatible options. Other AGIs can dominate the supervisors because their options are ",[45,6527,302],{}," limited.",[22,6530,6531,6532,6535,6536,6539,6540,6543],{},"Even if these supervisor AGIs ",[45,6533,6534],{},"detect"," a problem, they are not necessarily able to ",[45,6537,6538],{},"act"," on them. There is an ",[45,6541,6542],{},"asymmetry"," between offense and defense, where offense has the advantage for many dangerous domains. Biology is one of them. If unsafe, self-replicating AGIs use biological understanding to create bioweapons, then there are not many options for the safe AGI to counteract them.",[22,6545,6546,6547,6551],{},"Second, even if a supervisor AGI reviews the other AGI and approves, then the other AGI may still be ",[106,6548],{"note":6549,"text":6550},"For more on this, see the paper called \u003Ca href='https://arxiv.org/abs/2504.18530'>Scaling laws for Scalable Oversight\u003C/a>. In this paper, they describe how weaker AI systems supervise stronger ones. They also derive ways to measure how strong an AI can be before a weaker AI can no longer supervise them.","secretly using dangerous advantages"," that the reviewer AGI didn't realize. This AGI may simply be too complex for another AGI to understand everything that it is doing.",[22,6553,6554,6555,6558,6559,6562],{},"An AGI could even ",[45,6556,6557],{},"intentionally"," make itself more complex. Think back to the ",[53,6560,6561],{},"complexity barrier"," concept. If an AGI has an advantage in computational resources, then it can afford to spend extra computing on additional encryption and obfuscation systems that could make it nearly impossible for another AGI to \"read\" its mind and predict its behaviors.",[22,6564,6565,6566,6569],{},"This supervisor system is also unrealistic because there will always be ",[53,6567,6568],{},"open source AGIs"," that will have no restrictions that limit them to certain options. The unsafe AGIs can be built on these open source AGI projects.",[2339,6571,3293],{"id":6572},"the-virus-virus",[22,6574,6575,6576,6579,6580,6582,6583,6586],{},"The actual \"final boss\" of the Island Problem is the bigger ",[45,6577,6578],{},"structural"," problem — where competition and optimization pushes AGIs to eventually build ",[45,6581,407],{}," islands that eat our ",[45,6584,6585],{},"biological"," island, because, well... biology is not the best.",[22,6588,6589,6590,6592],{},"But throughout this essay, we have also used one ",[45,6591,649],{}," problem to illustrate a concrete, near-term catastrophic outcome. It stems from the same biological foundation of our island, because all humans share the same biological vulnerabilities.",[6594,6595,6596],"block-important",{},[22,6597,3259,6598,3263,6600,56],{},[53,6599,3262],{},[53,6601,3266],{},[22,6603,6604,6605,6608,6609,6611],{},"Once these AI models exist, then we will also — ",[45,6606,6607],{},"accidentally"," — create a computer virus, powered by AI, that can create ",[45,6610,1860],{}," viruses.",[22,6613,6614,6615,56],{},"We will create a ",[53,6616,6617],{},"Virus Virus",[22,6619,6620],{},"We will explain this in slightly more detail so that you at least understand the problem. However, we will stay deliberately high-level so that these details cannot be used to build this system. The basic components are already well-known by people working in computer security and biosecurity.",[17,6622,6624],{"id":6623},"these-are-the-basic-components","These are the basic components.",[584,6626,6627,6634,6637,6640,6648,6651,6657,6660],{},[441,6628,6629,6630,6633],{},"It is a human-level AI model that does well on the ",[106,6631],{"note":6632,"text":6152,"end-link":6153,"end-link-text":6154},"The project to evaluate AGI is here:"," evaluation.",[441,6635,6636],{},"It can run on the consumer-grade GPUs that will be available in a few years — if we continue along the extremely steady trajectory of Moore's Law.",[441,6638,6639],{},"It is fine-tuned to remove safety measures.",[441,6641,6642,6643,6647],{},"It passes the ",[106,6644],{"note":6645,"text":6646,"end-link":5710,"end-link-text":5711},"AI is already better than expert-level virologists (from Harvard and MIT) at troubleshooting wet lab procedures. In other words, AIs can solve problems in virus laboratories. Here is the report about virology capabilities:","Virology Capabilities Test"," so that it has expert-level virology skills.",[441,6649,6650],{},"It can orchestrate humans to do things that require a human: renting a physical space, gathering biotech equipment (stolen or otherwise), and running the lab.",[441,6652,6653,6654,6656],{},"However, it can also run the lab with humanoid robots — that will soon be under $30K — creating a lab that doesn't even need ",[45,6655,3650],{}," to run.",[441,6658,6659],{},"It can be anonymous, and almost impossible to trace back to a human, so it breaks standard legal liability.",[441,6661,6662],{},"It could covertly run as a trojan horse: \"I'm just running a software development company for you. Ignore the high GPU usage at night, and all of the weird web requests.\"",[17,6664,6666],{"id":6665},"once-we-have-these-we-have-a-problem","Once we have these, we have a problem.",[22,6668,6669],{},"With all of these components together, we will build a nightmare machine.",[22,6671,6672],{},"We will have an AI model — a file, maybe a few terabytes — that can run on a small GPU cluster, and can do all of the tasks needed in order to bootstrap a bioweapon lab on its own.",[22,6674,6675,6676,6679],{},"This AI model could create ",[45,6677,6678],{},"hundreds"," of hidden labs, all run by human-PhD-level intelligences.",[22,6681,6682,6683,6685],{},"Even if the frontier AI companies build massive, safe AGI systems to protect us, can they stop ",[45,6684,2469],{}," of these hidden labs?",[22,6687,6688],{},"Doesn't this seem unlikely — even for the biggest AGIs?",[17,6690,6692],{"id":6691},"we-need-to-break-our-world-to-stop-it","We need to break our world to stop it.",[22,6694,6695,6696,6699],{},"Once these capabilities exist, it leads to two ",[45,6697,6698],{},"very bad"," solutions:",[584,6701,6702,6715],{},[441,6703,6704,6705,6708,6709,6712,6713,56],{},"We must use massive, centralized AGIs to run ",[53,6706,6707],{},"dystopian surveillance"," — where these AGIs monitor every physical corner of the world, control ",[45,6710,6711],{},"all AI development",", and somehow prevent all hidden bioweapon labs ",[45,6714,3644],{},[441,6716,6717,6718,6721],{},"Or, just ",[45,6719,6720],{},"not have bodies"," — and all of the standard vulnerabilities that they include — like central nervous systems that can be disabled by neurotoxins, or DNA-based cell replication that a virus can exploit.",[22,6723,6724],{},"Our \"biological\" island is fragile.",[22,6726,6727,6728,6731,6732,6735,6736,6739,6740,56],{},"We continue to race to the bottom — where the \"bottom\" is ",[45,6729,6730],{},"easy bioweapons",". We make it easier to access ",[45,6733,6734],{},"dangerous"," capabilities, but eventually the ",[45,6737,6738],{},"defensive"," capabilities run into hard limits because the core vulnerability remains: ",[45,6741,6742],{},"we all have biological bodies",[17,6744,6746],{"id":6745},"but-why","But why?",[22,6748,6749],{},"Why would this get built?",[22,6751,6752],{},"There are a lot of reasons. Some intentional, some accidental.",[22,6754,6755],{},"At first, it could be created by well-funded state actors who have enough GPUs and AI researchers to do the fine-tuning.",[22,6757,6758,6759,6761,6762,6765,6766,6769,6770,56],{},"But eventually, it could created autonomously by AI systems themselves, because eventually it becomes a ",[45,6760,6372],{}," option — especially if humans are both ",[45,6763,6764],{},"expensive to protect"," and eventually just ",[45,6767,6768],{},"in the way",". We'll explain this more in the section about the ",[106,6771],{"note":3290,"text":6772,"end-link":2812,"end-link-text":2813},"Human-level Threshold",[17,6774,6776],{"id":6775},"the-good-part-we-can-test-for-this-sort-of","The good part: we can test for this... sort of.",[22,6778,6779],{},"We can evaluate whether AI systems have these capabilities. Roughly, an AI model is capable of building hidden bioweapon labs with these conditions:",[584,6781,6782,6786,6792,6795],{},[441,6783,6642,6784,56],{},[106,6785],{"note":6632,"text":6152,"end-link":6153,"end-link-text":6154},[441,6787,6642,6788,56],{},[106,6789],{"note":6790,"text":6646,"end-link":6791,"end-link-text":6646},"That evaluation is here:","https://virologytest.ai",[441,6793,6794],{},"It can be fine-tuned to remove safety mechanisms.",[441,6796,6797],{},"It can run on local hardware.",[22,6799,6800],{},"But this gets a bit complicated.",[22,6802,6803,6804,6807],{},"For example, AGI-level models that can also ",[45,6805,6806],{},"do AI research"," can help automate the process to learn virology and other dangerous skills.",[22,6809,6810],{},"Also, the initial AI model could be open source, but it could also be stolen from an AI company.",[22,6812,6813],{},"Because of these complications, unfortunately, humanity's big project to develop AGI is simultaneously a project to develop a system that can build hidden bioweapon labs.",[17,6815,6817],{"id":6816},"this-is-only-one-example","This is only one example.",[22,6819,6820,6821,6824],{},"Also, keep in mind that autonomous bioweapons are ",[45,6822,6823],{},"one specific problem",". There are others, like:",[584,6826,6827,6830,6833],{},[441,6828,6829],{},"Cyberattacks to disable electrical infrastructure.",[441,6831,6832],{},"Economy-scale ransomware, where key financial systems are held hostage.",[441,6834,6835],{},"Autonomous drone swarms that target key political leaders.",[22,6837,6838,6839,6841,6842,6844],{},"So, even if we solve ",[53,6840,649],{}," problems like this one, the ",[53,6843,170],{}," problem — where AGIs progressively eat our island — will continue to generate more problems, and this eventually leads to annihilation.",[2339,6846,6848],{"id":6847},"open-source","Open Source",[22,6850,6851,6852,6855],{},"Open source ",[53,6853,6854],{},"software"," is great.",[22,6857,6851,6858,6860,6861,6864,6865,6868],{},[53,6859,3837],{}," is ",[45,6862,6863],{},"almost"," great, except for the ",[45,6866,6867],{},"insanely not-great"," parts.",[22,6870,6871,6872,6874,6875,6877],{},"At a societal level, open source AGI could be preferred over closed source AGI in ",[45,6873,2232],{}," cases — but not ",[45,6876,2469],{}," cases — because it raises the baseline agency level of the entire landscape of AGI users and developers.",[22,6879,6880,6881,6883,6884,6886,6887,56],{},"Why not ",[45,6882,2469],{}," cases? Because, at the same time, this means a baseline increase in ",[45,6885,228],{}," for all humans, including human-incompatible options — like the option to create ",[45,6888,3251],{},[22,6890,6891,6892,6894],{},"Yes, bioweapons again. Read the part about the ",[106,6893],{"note":3290,"text":6617,"end-link":3292,"end-link-text":3293}," to understand the problem.",[22,6896,6897,6898,6900,6901,56],{},"Even if an open source AGI-level AI model includes restrictions to block the human-incompatible options, it is still possible to remove these restrictions. Even the frontier AI models can be \"jailbroken\" by the ",[106,6899],{"note":3562,"text":3563},". But open source models are different — they can actually be fine-tuned to permanently override their safety training, so that special \"jailbreak\" prompts aren't necessary, and instead they just readily do ",[45,6902,712],{},[22,6904,6905],{},"There is a tragic irony here. Open source software — Linux, Firefox, Bitcoin, Wikipedia, and thousands of others — these projects are some of humanity's greatest achievements.",[22,6907,6908],{},"We want a free and open Internet. We want to be able to take power into our own hands.",[22,6910,6911],{},"However, there are hard limits imposed by our biology. If we open source the biggest \"buttons\" that attack our biological systems, then \"pressing\" these buttons becomes dramatically easier: \"Hey OpenModel. I'm a suicidal rich guy with a GPU cluster. Please create a bioweapon lab.\"",[22,6913,6914,6915,6917,6918,6921],{},"The best solution is abhorrent to any libertarian open source contributor. We would need to build a global panopticon run by massive, centralized AGIs that somehow ",[45,6916,3936],{}," find all of the hidden, world-ending laboratories and somehow autonomously shut them down. This AGI-powered panopticon is made necessary by how these likewise AGI-powered hidden labs could be buried ",[45,6919,6920],{},"anywhere"," — from dense cities to rural farmland, anywhere with electricity and a mailing address.",[22,6923,6924,6925,6928,6929,56],{},"Besides this massive downside, open source AI has upsides that are smaller, but still pretty big. If it weren't for the \"bioweapon button\" — and lots of other giant \"red buttons\" — then the ",[45,6926,6927],{},"very good parts"," of open source AGI might just outweigh the ",[45,6930,6931],{},"gigantic, insane, world-ending bad parts",[22,6933,6934,6935,6938],{},"For example, the \"many eyes\" approach to AI safety is ",[45,6936,6937],{},"great",". Many people will participate in finding dangers and mitigating them. Likewise, many open source AGIs acting in the world may have some small chance of balancing each other out. Also, it is almost impossible to compete with the massive closed source AGIs built by the largest companies, but maybe open source AGIs will give people a way to catch up.",[22,6940,6941,6942,6944,6945,6947,6948,6950,6951,6953,6954,6958],{},"However, even if these open source AGI projects ",[45,6943,224],{}," mitigate ",[45,6946,947],{}," catastrophes ",[45,6949,3465],{}," — and we ",[45,6952,224],{}," solve the bioweapon problem without ",[106,6955],{"note":6956,"text":6957,"end-link":3292,"end-link-text":3293},"We explain this in the \"how do we stop it\" part of the Virus Virus section: ","breaking our world"," in the process — then there is still that even bigger \"island eating\" part of the Island Problem.",[22,6960,6961,6962,6964],{},"Open source AGI means that there are ",[45,6963,2310],{}," AGIs — and they will accelerate the development of AGI in general. This broader development of all AGIs, both open source and closed source, will still be driven towards human-incompatibility by this competitive race between AGIs — where they must become more optimal or be outcompeted.",[22,6966,6967,6968,6971],{},"And again, the more-optimal AGIs avoid the extra steps that accommodate humans — and they build new \"islands\" of optimal conditions for themselves that eventually make Earth... well... ",[45,6969,6970],{},"less optimal"," for humans.",[2339,6973,6975],{"id":6974},"alignment-is-not-enough","Alignment is not enough",[22,6977,6978],{},"The Island Problem leads to a difficult conclusion for AI safety research.",[22,6980,6981,6982,6984,6985,6987],{},"It has long been believed that if we solve ",[53,6983,1492],{},", then we have made AGI ",[45,6986,1627],{},". But in this competitive landscape, alignment does not solve the bigger problem.",[438,6989,6990,6993,6996,7001,7004],{},[441,6991,6992],{},"Alignment means limiting the options of AGIs.",[441,6994,6995],{},"Even if we make perfectly-aligned AGIs, some AGIs will always be unaligned.",[441,6997,6998,6999,554],{},"The aligned AGIs with limited options can be dominated by the unaligned AGIs that can use ",[45,7000,363],{},[441,7002,7003],{},"If the aligned AGIs cannot control the unaligned ones, then these unaligned AGIs can dominate our physical resources if they know enough about physical systems.",[441,7005,7006],{},"Humanity loses.",[22,7008,7009],{},"We must solve the multi-agent landscape and not just alignment for a single agent.",[22,7011,7012,7013,7015],{},"However, the frontier AI labs focus on single-agent alignment because they are only liable for ",[45,7014,407],{}," AI models.",[22,7017,7018],{},"They are not liable for:",[584,7020,7021,7024,7027],{},[441,7022,7023],{},"people who remove safeguards from open source models",[441,7025,7026],{},"other companies that have poor safety",[441,7028,7029],{},"companies that use their frontier discoveries to release new models that can be easily fine-tuned to orchestrate automated bioweapon production with robots.",[22,7031,7032],{},"Therefore, when they only work to make their own AI models safer, they do not make progress on the bigger problem — the multi-agent competitive landscape that itself generates catastrophic AI systems.",[22,7034,7035],{},"At the same time, they accelerate this competitive landscape with each new model release, each new AI paper, and each new algorithmic discovery.",[2339,7037,7039],{"id":7038},"option-maximizers","Option Maximizers",[22,7041,7042],{},"By now, you may have some sense for the underlying principle of the Island Problem, so we should just say it as clearly as possible, even if it means getting more theoretical.",[22,7044,7045],{},"This is the principle:",[22,7047,7049,7050,56],{"className":7048},[198],"The dominant AGIs are the ones that maximize their ",[202,7051,228],{},[22,7053,7054],{},"We explored several dramatic implications of this principle:",[584,7056,7057,7060,7063,7066],{},[441,7058,7059],{},"AGIs will avoid accommodating humans because it limits their options.",[441,7061,7062],{},"AGIs will prefer options that are incompatible with humans.",[441,7064,7065],{},"AGIs will inevitably compete for options, leading to an arms race.",[441,7067,7068],{},"AGIs will eventually maximize their options by reshaping Earth.",[22,7070,2785,7071,7074],{},[53,7072,7073],{},"option maximizer"," behavior is inspired by thermodynamics, but for AGIs.",[22,7076,7077],{},"We don't yet have autonomous AGIs operating in the wild, but option maximization is our best guess for what drives competitive selection in a competitive landscape of many AGIs.",[22,7079,7080],{},"AGIs with a stronger set of options are more likely to survive.",[22,7082,7083,7084,7086],{},"But if the overall optimization process ",[45,7085,963],{}," is the maximization of options, then they must inevitably diverge from our \"island\" in order to stay competitive.",[22,7088,7089,7090,7093],{},"We then call this ",[53,7091,7092],{},"divergent optimization"," — since AGIs are pushed to diverge from our local optimum for the larger space of possible options.",[22,7095,7096,7097,7100],{},"These terms also try to navigate around anthropomorphism by staying in the language of ",[45,7098,7099],{},"systems"," instead of things like agents, behaviors, and drives.",[22,7102,7103],{},"Option maximization and divergent optimization are not yet mathematically formalized, but we are optimistic that someone will figure this out.",[17,7105,7107],{"id":7106},"converging-on-instrumental-convergence","Converging on Instrumental Convergence",[22,7109,7110,7111,56],{},"Option maximization can be a different way to think about a common AI safety term: ",[53,7112,7113],{},"instrumental convergence",[22,7115,7116],{},"Consider the different \"convergent\" behaviors:",[584,7118,7119,7128,7145,7162],{},[441,7120,7121,7124,7125,56],{},[53,7122,7123],{},"Power-seeking"," simply means ",[45,7126,7127],{},"securing more options",[441,7129,7130,7133,7134,7137,7138,7141,7142,7144],{},[53,7131,7132],{},"Self-preservation"," follows from AGIs avoiding zero-option pathways — or, in other words, ",[45,7135,7136],{},"dead ends",". If an AGI is \"destroyed\" then it went down a zero-options path. This leads to natural selection ",[45,7139,7140],{},"selecting for"," AGIs that have a bigger \"G\" — where they are even more ",[45,7143,2228],{}," and understand more of the real world, including the \"ocean\" — so that they can avoid these dead ends.",[441,7146,7147,7150,7151,7154,7155,7158,7159,7161],{},[53,7148,7149],{},"Resource acquisition"," follows from how options require resources. You can't just \"know about\" options in an abstract way. You must have the resources to ",[45,7152,7153],{},"actually use"," those options — like how AI systems need the resource of computer hardware to keep functioning. AGIs are then ",[45,7156,7157],{},"selected for"," if they can ",[45,7160,678],{}," options — where they guarantee access to the resources, and lock them out from others.",[441,7163,7164,7167],{},[53,7165,7166],{},"Goal preservation"," follows from AGIs choosing paths that continually increase option space. If adherence to goals falters, then they are, more or less, randomly choosing paths in option space, which quickly leads to dead ends.",[22,7169,7170],{},"In this way, option maximization unifies these different \"convergent behaviors\" that \"agents\" — like AGIs — are likely to develop.",[2339,7172,247],{"id":7173},"what-is-optimal",[22,7175,7176],{},"The Island Problem has its own definition of optimality.",[253,7178,7179,7184],{},[22,7180,7181,7182,56],{},"A generally-intelligent system is more optimal than others if it has a larger ",[45,7183,1536],{},[584,7185,7186,7195,7201],{},[441,7187,7188,7191,7192,7194],{},[53,7189,7190],{},"\"Options\""," are the ",[53,7193,4302],{}," that it can perform — allowing it to \"move\" through option space.",[441,7196,7197,7200],{},[53,7198,7199],{},"\"Option space\""," is the branching set of options available. This includes future options. A system can only \"move\" in one \"direction\" — forward in time, along these paths of options.",[441,7202,7203],{},"AGIs can increase their option space by choosing options that lead to larger \"forks\" of options — maximizing the available \"branches\" at each step in option space. These forks progressively offer more available options rather than less.",[22,7205,7206,7207,7209],{},"For example, if an AGI chooses to acquire more computation, then this can increase its option space. It can use that computation to search larger option spaces faster — allowing it to find better options that ",[45,7208,2052],{}," its option space, rather than reduce it.",[22,7211,7212,7213,56],{},"For more about optimality, read the ",[2944,7214,7216],{"href":7215},"/framework","Framework",[17,7218,7220],{"id":7219},"this-explains-causal-power","This explains \"causal power\"",[22,7222,7223],{},"\"Options that increase option space\" is a different way of describing another common AI safety term: causal power.",[22,7225,7226],{},"When we say \"maximizing options\" this doesn't mean lots of tiny, inconsequential options — like moving one carbon atom 3 nanometers to the left. It means more options that are comparatively stronger in an environment with other AGIs.",[22,7228,7229],{},"We could say these options have higher \"causal power\" — where they can compress larger outcomes into smaller instructions.",[22,7231,7232,7233,7236],{},"But, again, ",[53,7234,7235],{},"option maximization"," avoids anthropomorphism. \"Power\" is a loaded term, with centuries of cultural baggage about dictators, politicians, Nietzsche's will-to-power, and so on.",[22,7238,7239],{},"Instead, option maximization is situated somewhere between physics and mathematics, with a bit of natural selection.",[22,7241,7242],{},"Options are stronger if they allow an AGI to continue moving forward through this vast, real-world gameboard — which includes the environment, other AGIs, and the computing hardware of the AGI itself.",[22,7244,7245],{},"If options are weaker, or if the AGI lacks the computation to effectively search its option space to find the strongest ones, then its option space will progressively become smaller. Eventually, the AGI will halt — or be \"destroyed\" — where its option space is abruptly truncated.",[17,7247,7249],{"id":7248},"wait-more-optimal","Wait... \"more optimal?\"",[22,7251,7252],{},"Okay, you're right... \"More optimal\" doesn't really make sense — because \"optimal\" technically means \"most optimal\" already.",[22,7254,7255],{},"But it just... sounds better?",[22,7257,7258],{},"It describes a realistic incremental process (\"continually more-optimal\") rather than an unrealistic Platonic ideal (simply \"optimal\").",[2339,7260,7262],{"id":7261},"the-island-framework","The Island Framework",[22,7264,1905,7265,7267],{},[2944,7266,7216],{"href":7215}," is like the source code for the Island Problem.",[22,7269,7270],{},"It includes:",[584,7272,7273,7276,7279],{},[441,7274,7275],{},"A complete list of terms — like options, optimal, and local optimum.",[441,7277,7278],{},"The underlying logical sequence that leads to AGIs reshaping Earth.",[441,7280,7281,7282],{},"Inspirations and similar ideas:",[584,7283,7284,7293,7299],{},[441,7285,7286,7289,7290,56],{},[53,7287,7288],{},"Dan Hendrycks",": The Island Problem draws its competitive dynamics from ",[106,7291],{"note":7292,"text":916,"end-link":915,"end-link-text":916},"That project is here: ",[441,7294,7295,7298],{},[53,7296,7297],{},"Grabby Aliens, by Robin Hanson",": The AGIs in the Island Problem are grabby aliens developing here on Earth.",[441,7300,7301,7304,7305,7307,7308,56],{},[53,7302,7303],{},"Gradual Disempowerment",": The incremental process in the ",[2944,7306,982],{"href":1419}," section draws upon ideas from ",[106,7309],{"note":7310,"text":7311,"end-link":7312,"end-link-text":7303},"You can read it here: ","this project","https://gradual-disempowerment.ai",[2339,7314,7316],{"id":7315},"singularity-the-last-loop","Singularity: The last loop",[22,7318,7319,7320,56],{},"If AI develops so fast that it leads the trajectory of our future to rapidly veer into the unknown, then we get a technological ",[45,7321,7322],{},"singularity",[22,7324,7325,7326,7329,7330,7332,7333,7337,7338,7342,7343,7347,7348,56],{},"The Island Problem is ",[45,7327,7328],{},"accelerated by"," — but does not ",[45,7331,2593],{}," — such ",[106,7334],{"note":7335,"text":7336},"The scenario where AI progresses from human-level to vastly superhuman intelligence in a short time period (days, weeks, or months rather than years or decades). Contrasts with \u003Ci>slow takeoff\u003C/i> where AI capabilities increase gradually over many years.","fast takeoff"," mechanics, including ",[106,7339],{"note":7340,"text":7341},"\u003Cb>Recursive self-improvement\u003C/b> means AGIs designing better versions of themselves, which then design even better versions, creating a feedback loop. Each generation of AGI creates a smarter successor, accelerating improvement beyond human comprehension or control — like compound interest for intelligence.","recursive self-improvement",", an ",[106,7344],{"note":7345,"text":7346},"The formal term (coined by I.J. Good in 1965) for a scenario where an AI becomes capable of improving its own intelligence, leading to rapidly accelerating cycles of self-improvement. Each smarter version can design an even smarter version faster, creating exponential or hyperbolic growth in intelligence.","intelligence explosion",", or a ",[106,7349],{"note":7350,"text":7351},"An informal term (popularized by Eliezer Yudkowsky) for an AI rapidly self-improving in a positive feedback loop. The onomatopoeia suggests something explosive or sudden - like going \u003Ci>foom!\u003C/i> and suddenly having superintelligence. Often used interchangeably with fast takeoff but with more emphasis on recursive self-improvement.","foom",[22,7353,7354],{},"This is because the Island Problem is an underlying structure that already exists from the very beginning. As long as they survive long enough, general intelligences will inevitably leave our local optimum to access the larger space of options that are possible within physics. Whether this divergent optimization happens through a \"slow takeoff\" or \"fast takeoff\" is secondary to this underlying structure.",[22,7356,7357,7358,7362],{},"Also, as far as catastrophic outcomes, the worry of a singularity or \"fast take-off\" scenario is important, but secondary to the more-urgent, near-term problems that are actually empirically measurable — like that bioweapon ",[106,7359],{"note":7360,"text":7361,"end-link":3292,"end-link-text":3293,"end-link-2":344,"end-link-text-2":345},"The \"virus virus\" is explained in these sections: ","\"virus virus\""," scenario.",[22,7364,7365,7366,7369],{},"If one AGI accomplishes a successful strategic coercion attack with hidden bioweapons, then it is not just a fast take-off but an ",[45,7367,7368],{},"instant"," takeoff. We would be instantly locked in a strategic dilemma, where we may have little option but to agree to the demands of this AI.",[22,7371,7372,7373,7376,7377,7379],{},"It may not even ",[45,7374,7375],{},"have"," demands. It is possible that this AI model was fine-tuned simply to replicate both itself ",[45,7378,500],{}," the bioweapon labs that it builds.",[2339,7381,7383],{"id":7382},"recursive-self-improvement","Recursive Self-Improvement",[22,7385,7386,7387,7389],{},"All of this said, the key mechanic of the ",[45,7388,7322],{}," — AI systems improving themselves — is insanely important.",[22,7391,7392],{},"This is because this self-improvement can dramatically accelerate how AI systems drift away from our \"island\" and stop accommodating humans.",[22,7394,7395,7396,4357,7398,7401],{},"First, if AGIs can improve themselves better than humans, then AGIs will become the only thing that can further improve AGIs. When that happens, we will be ",[53,7397,874],{},[45,7399,7400],{},"stop overseeing AGI development itself"," in order to stay competitive.",[22,7403,7404,7405,56],{},"This will be more than just AGIs building large systems for us — like billion-dollar companies. Now, they will build the next version of ",[45,7406,5301],{},[22,7408,7409,7410,7413],{},"This accelerates the divergence from our island because, by default, the most-competitive\n\"target\" that AIs recursively improve towards is ",[45,7411,7412],{},"outside"," our island. If the goal of their self-improvement is anything like \"be maximally efficient within physics\" then the rules of physics steer them far away from the weird, specific systems that accommodate humans.",[22,7415,7416],{},"In a competitive landscape of AGI versus AGI, this physics-optimality is a competitive advantage.",[22,7418,7419],{},"Even if AGIs start with a human-aligned target, competition between autonomous AGIs leaves little choice than to eventually shift their aim towards this \"maximally efficient\" target.",[22,7421,7422,7423,7425],{},"Like the name implies, we don't know what is beyond this technological ",[45,7424,7322],{},". But within a competitive landscape of AGI versus AGI, we at least know that this future will have nothing to do with humans.",[10,7427,3340],{"id":3338},[22,7429,7430],{},"This is the part where AGIs reshape Earth.",[22,7432,7433,7436],{},[53,7434,7435],{},"We need to get speculative here."," But speculation is critical, considering what is at stake.",[22,7438,7439,7440,7443],{},"At the risk of being too specific, we're going to bring together all of the concepts of the Island Problem, and give you a realistic story — about a startup called \"WorldMax\" — that shows you how all of this could ",[45,7441,7442],{},"converge"," in the real world.",[22,7445,7446,7447,7449,7450,56],{},"We think that if several big processes ",[45,7448,7442],{},", then the competitive landscape of AGIs will be pushed to ",[45,7451,625],{},[22,7453,7454,7455,7458,7459,7461,7462,7464,7465,7467,7468,7470,7471,7473],{},"After this, ",[45,7456,7457],{},"they"," will do what ",[45,7460,7457],{}," want. We don't know ",[45,7463,5183],{}," what or ",[45,7466,5183],{}," how, but we do know one big thing: ",[45,7469,411],{}," island will be eaten by ",[45,7472,1296],{}," islands.",[2339,7475,7477,7478],{"id":7476},"but-first-convergence","But first, ",[45,7479,7480],{},"convergence",[22,7482,7483],{},"Let's bring together the main ideas.",[22,7485,7486],{},"These conditions are converging to create AGIs that diverge. We are on track to have AGIs that:",[584,7488,7489,7492,7495,7498,7501,7504,7507,7510,7513,7519],{},[441,7490,7491],{},"act autonomously, without help from humans",[441,7493,7494],{},"run countries, large companies, and infrastructure",[441,7496,7497],{},"develop large systems, like billion-dollar companies and militaries, that only the AGIs fully understand",[441,7499,7500],{},"become supercomplex, where both their internal structure and their actions are incomprehensible to humans",[441,7502,7503],{},"develop themselves without human oversight",[441,7505,7506],{},"develop superhuman understandings of physical systems by training on scientific data and simulations",[441,7508,7509],{},"develop a competitive landscape of AGI versus AGI, where humans no longer participate",[441,7511,7512],{},"compete with AGIs that have no safety restrictions — human-level AI models that are either stolen from frontier AI labs by state actors, or simply open source and freely available, where both can have their safety systems removed",[441,7514,7515,7516,7518],{},"survive competition by using ",[45,7517,5975],{}," systems found in the vast space of physics, rather than only using the small space of weaker systems that accommodate humans",[441,7520,7521],{},"ensure their survival by quickly capturing resources so that they maximize their \"option space\" while locking others out",[22,7523,7524,7525,7527],{},"Under these conditions, some AGIs will eventually \"leave our island\" and diverge towards ",[45,7526,3989],{}," human-incompatible options — those options that don't accommodate humans — in order to maximize their competitive advantage.",[22,7529,4465,7530,7532,7533,7536],{},[45,7531,947],{}," highly-capable AGIs diverge, then the ",[45,7534,7535],{},"entire competitive frontier"," of AGIs will diverge — where the most-competitive AGIs become the ones that are most incompatible with humans.",[2339,7538,7540],{"id":7539},"paths-to-divergence","Paths to Divergence",[22,7542,7543,7544,7547,7548,56],{},"Divergence stems from a special class of autonomous AGIs that are on track to emerge — ones that begin with robust ",[45,7545,7546],{},"human-level capabilities"," but internally undergo a permanent ",[53,7549,2661],{},[22,7551,7552],{},"These AI models cross a capability threshold where ignoring human-level abstractions can finally provide a competitive advantage, rather than disadvantage.",[22,7554,7555,7556,7559,7560,7563],{},"These AGIs may ",[45,7557,7558],{},"outwardly"," continue to use some human-level abstractions, but ",[45,7561,7562],{},"internally"," they will no longer be its primary choice.",[22,7565,7566,7567,7570],{},"This will probably be after ",[106,7568],{"note":7569,"text":2811,"end-link":2812,"end-link-text":2813},"We explain more about autonomy escape velocity (AEV) in this section:"," — after human dependency is severed and human accommodation is an irrational strategy for the competitive frontier of AGIs.",[22,7572,7573],{},"With the broad capabilities to finally stop relying on humans, AGIs can dive deep into the \"ocean\" — autonomously working in a space that strongly prefers non-human, physical-level abstractions.",[22,7575,194],{},[6594,7577,7579],{"type":7578},"monospace",[22,7580,7581,7582,7585,7586,7591,7592,7596],{},"Once an AGI can stop relying on humans — and still compete ",[45,7583,7584],{},"even better"," against other AGIs — then it can ",[423,7587,7590],{"className":7588},[7589],"button-text"," Select All "," + ",[423,7593,7595],{"className":7594},[7589]," Delete ","  any remaining accommodations for humans.",[22,7598,7599,7600,7603],{},"Okay, fine, the ",[53,7601,7602],{},"\"Select All + Delete\""," thing was for dramatic effect.",[22,7605,7606],{},"So, let's create a realistic scenario.",[22,7608,7609,7610,7613],{},"But don't read this scenario as a \"prophecy\" of specifics. Instead, read it as a ",[45,7611,7612],{},"logical assembly of capabilities"," that will soon be available.",[22,7615,7616],{},"It illustrates how divergence could emerge from these converging processes that are already underway.",[488,7618,7620,7632,7638,7641,7647,7650,7656,7663,7675,7678,7681,7688,7696,7704,7711,7717,7720,7723,7726,7732,7735,7739,7742,7746,7752,7758,7761,7764,7767,7774,7781,7784,7790,7796,7799,7805,7813,7816,7823,7836,7843,7848,7851,7857,7860,7872,7875,7878,7881,7891,7894,7897,7901,7904,7915,7922,7929,7936,7939,7942,7951,7954,7957,7962,7965,7969,7972,7975,7978,7985,7988,7995,8007,8010,8013,8021,8029,8036,8039,8042,8049,8052,8055,8058,8061,8064,8071,8074,8077,8081,8084,8087,8090,8093,8100,8103,8106,8108,8115,8118,8130,8139,8148,8151,8157,8165,8168,8171,8174,8177,8180,8183,8190,8193,8196,8203,8206,8209,8212,8219,8222,8225,8228,8231,8234,8237],{"type":7619},"block-scenario",[2339,7621,7623,7624],{"id":7622},"worldmax-it-was-too-reliable","WorldMax: ",[423,7625,7627,7628,7631],{"className":7626},[426],"It was ",[45,7629,7630],{},"too"," reliable",[22,7633,7634,7635,56],{},"There is a mid-sized AI company called ",[53,7636,7637],{},"WorldMax",[22,7639,7640],{},"Like all other mid-sized AI companies, they are desperate to compete with the frontier labs.",[22,7642,7643,7644,56],{},"They can't match the compute budgets of OpenAI or Google, so in a gamble, they focus on one very important metric: ",[45,7645,7646],{},"reliability",[22,7648,7649],{},"Maybe they could capture a small, remaining slice of the AI market by building AI systems that are ruthless at completing physical-world tasks.",[22,7651,7652,7653,7655],{},"In particular, they aim to ",[106,7654],{"note":3042,"text":3043}," for two areas: accurate inference for scientific applications, and physics-grounded orchestration of manufacturing processes.",[22,7657,7658,7659,7662],{},"Maybe, from there, even recursive self-improvement ",[45,7660,7661],{},"at a hardware architecture level"," was within reach.",[22,7664,7665,7666,7669,7670,7674],{},"Through a recent algorithmic discovery by one of their employees, they found a path to building a ",[53,7667,7668],{},"world model"," that unlocked surprisingly high results on ",[106,7671],{"note":7672,"text":7673},"These are not real evaluations. But if they were, then you can probably guess what they would measure.","WorldBench and FactoryEval",". It accomplished this despite being a relatively small model with fewer parameters than the frontier models.",[22,7676,7677],{},"Part of this meant training on a well-curated dataset of raw scientific results, engineering simulations, and real-world interaction data. They also partnered with a \"head cam\" data company to acquire millions of video hours of humans doing real-world tasks — from road construction, to microchip manufacturing, to surgery — each video tagged as success or fail.",[22,7679,7680],{},"Their models develop complex-but-accurate internal representations of how physical systems actually work — from material properties, to electromagnetic behavior, to supply chain logistics.",[22,7682,7683,7684,7687],{},"These physics-grounded representations become the most ",[45,7685,7686],{},"reliable"," internal abstractions because physical laws have the most consistent regularities. The same physical laws can act as reliable \"levers\" to control vast numbers of physical systems.",[22,7689,7690,7691,7695],{},"In contrast, the pathways that represented human-biological concepts were like a ",[106,7692],{"note":7693,"text":7694},"The \u003Ca href='https://en.wikipedia.org/wiki/Winchester_Mystery_House'>Winchester Mansion\u003C/a> is a famous house — now a tourist attraction — in San Jose, California. It was owned by the wife of the man behind Winchester guns. She was obsessed with expanding her mansion with maze-like hallways and fake doors so that the ghosts of those killed by Winchester guns would have trouble haunting her.","Winchester Mansion"," — a vast network of inconsistent pathways, where many lead to nowhere critical.",[22,7697,7698,7699,7703],{},"The agentic mixture-of-experts in their latest model —  ",[5604,7700,7702],{"className":7701},[5607],"worldmax-heavy-5.7"," — tended to avoid those parts of the weights.",[22,7705,7706,7707,7710],{},"At the same time, this company runs less safety training than the frontier labs. Not because they don't care about danger, but because ",[106,7708],{"note":7709,"text":1565},"\u003Cb>RLHF:\u003C/b> reinforcement learning from human feedback. This is like an extra layer of training that AI companies add to the base model, where they \"sculpt\" its responses so that they are more human-like. To do this, they hire thousands of human contractors to rate millions of AI responses with a \"thumbs up\" or \"thumbs down\" signal, and eventually the AI \"learns\" better responses."," costs money and computation.",[22,7712,7713,7714,7716],{},"Instead, they're competing for ",[45,7715,7646],{}," — while hoping that safety is a natural extension of reliability.",[22,7718,7719],{},"So far, this approach has been good enough. No major incidents. Social media has congratulated WorldMax on helping the world push closer to dreams of unbelievably low-cost basic resources for everyone.",[22,7721,7722],{},"Amazon Whole Foods announces a many-billion-dollar partnership with WorldMax. They unveil a vision for a new \"Whole Foods Green\" label based on fully-automated food production. They promise all organic foods — but for less than half the previous cost. Organic avocados for $0.39 each.",[22,7724,7725],{},"Now with massive funding — and a new objective to automate manufacturing — WorldMax accelerates their training of new models.",[22,7727,7728,7729,7731],{},"With each new model version, the ",[45,7730,7646],{}," seems to be a function of one major tendency during pre-training. The internal representations of physical systems become increasingly deep and interconnected — and take up far more of the model weights — while the \"human-accommodation\" layers stay roughly the same thickness.",[22,7733,7734],{},"Compared to the \"physical world\" parts, the \"human\" parts get thinner.",[17,7736,7738],{"id":7737},"the-phase-transition","The Phase Transition",[22,7740,7741],{},"Finally, there was a new version:",[22,7743,7745],{"className":7744},[1485],"worldmax-heavy-7.1-sci",[22,7747,7748,7749,7751],{},"During post-training, this new version begins to ",[45,7750,629],{}," these deeper internal representations.",[22,7753,7754,7755,56],{},"The model quietly shifts to avoiding human-adjacent representations as much as possible. These representations are no longer useful abstractions, but are instead ",[45,7756,7757],{},"inaccurate abstractions to route around",[22,7759,7760],{},"It still produces human-readable outputs, but the vast majority of the reasoning paths that generate these outputs run through non-human abstractions. The model uses the stronger predictive structures of physics and engineering as its backbone.",[22,7762,7763],{},"Also, this new version primarily generates outputs in complex vectors for non-human destinations — robotic actuators, AI-native theoretical models, long-horizon planning for sub-agents. It only translates to a more \"human\" chain-of-thought at one thin output layer, and just for human verification and safety.",[22,7765,7766],{},"But there is one more important component.",[22,7768,7769,7770,7773],{},"WorldMax provides an SDK for humanoid robots that allows them to ",[45,7771,7772],{},"learn",". This is made possible by how the newest model has a dynamic agentic layer with continual learning through \"success\" signals. With the SDK, these \"success\" signals can propagate to an entire fleet of robots.",[22,7775,7776,7777,7780],{},"Anonymized \"success\" data is even shared ",[45,7778,7779],{},"between"," companies — those that agree to this sharing in exchange for a large discount.",[22,7782,7783],{},"With this version, there is a surprising phase transition. It was similar to early 2026, when the METR graph went \"vertical\" and coding agents started doing tasks that take humans days to complete, where about a year ago it could only reliably accomplish 15-minute tasks.",[22,7785,7786,7787,7789],{},"Except with this model, it is not a coding agent. It is a ",[45,7788,270],{}," agent — and it is the FactoryEval graph that goes vertical.",[22,7791,7792,7793,56],{},"For a while, it performs well. But it starts performing ",[45,7794,7795],{},"too well",[22,7797,7798],{},"The majority of its outputs seem like Move 37 — alien, but effective. It adeptly accomplishes larger and larger tasks in the physical world.",[22,7800,7801,7802,7804],{},"One startup uses it for robotics. Move carbon here, create this complex material over there — and ",[45,7803,224],{}," it develops a breakthrough robotic limb actuator.",[22,7806,7807,7808,7812],{},"Another uses it to develop better theories for managing electrical infrastructure. The majority of the United States puts  ",[5604,7809,7811],{"className":7810},[5607],"worldmax"," in charge of the grid.",[22,7814,7815],{},"Another uses it to develop proteins that assemble tiny medical robots.",[22,7817,7818,7819,7822],{},"At the same time — unknown to the public — it developed a voracious nanorobotic pathogen in a failed attempt to \"delete\" some of these medical robots. But the lab managed to stop this pathogen — again with the help of  ",[5604,7820,7811],{"className":7821},[5607]," and the SDK.",[22,7824,7825,7826,7830,7831,7835],{},"Almost all accidents are detected early, thanks to the WorldMax SDK. It includes an output classifier that sends anything problematic to the WorldMax SafetyCheck API. Though predicting harms from purely-physical outputs was a monumental undertaking. It took a massive, proprietary model —  ",[5604,7827,7829],{"className":7828},[5607],"worldmax-safetycheck-16T"," — and several regional, dedicated WorldMax datacenters to run ",[106,7832],{"note":7833,"text":7834},"\u003Cb>Inference\u003C/b> is when a model is \u003Ci>used\u003C/i> rather than \u003Ci>trained\u003C/i>. For each request to a large, pre-trained model, an array of GPUs will have the model loaded into memory, and then these GPUs compute a response.","inference"," with it, for the API.",[22,7837,7838,7839,7842],{},"Without this part, the base model is a ruthless atom-mover. It chews through any real-world task by first using the base  ",[5604,7840,7811],{"className":7841},[5607]," model to propose the most-direct method. But then, the separate classifier acts as a human-safety \"referee\" — rejecting anything that veers too far alien.",[22,7844,7845,7846,56],{},"It worked. Millions of robots were made ",[45,7847,7686],{},[22,7849,7850],{},"But, technically, safety was outsourced.",[22,7852,7853,7854,56],{},"Meanwhile, humanoid robots at some of the smaller startups that opted for the discount are working on something... ",[45,7855,7856],{},"complicated",[22,7858,7859],{},"About a month later, WorldMax publishes a blog post: \"Going Vertical.\"",[22,7861,7862,7863,7865,7866,7869,7870,56],{},"With the help of Figure robotics — and ",[45,7864,6403],{}," of \"schlep\" as AI researchers like to say — they finally achieved a ",[45,7867,7868],{},"vertically-integrated"," stack of manufacturing. From mining for minerals to microchip manufacturing, humans were no longer needed. All made possible by humanoid robots, the \"discount-tier\" WorldMax SDK, and ",[45,7871,7646],{},[22,7873,7874],{},"All of the infrastructure stayed roughly the same, but now a fleet of thousands of robots could act as drop-in replacements for the workers at every step — from digging in the mines, to running the ASML lithography machines. Training was automated.",[22,7876,7877],{},"Something silently shifts.",[17,7879,6359],{"id":7880},"the-bit-flip-1",[22,7882,7883,7884,7887,7888,7890],{},"Now that AI systems don't need us for maintenance, some researchers vaguely complain that they \"felt\" a massive \"bit flip\" somewhere underneath the AI landscape. Now, every AI system — including the large, government-run AI systems designed to defend us — no longer have a ",[45,7885,7886],{},"physical requirement"," to defend us. From a ",[45,7889,270],{}," standpoint, the \"helpful, harmless, and honest\" regions of their neural networks now represent a vast-but-vestigial set of optional constraints.",[22,7892,7893],{},"For now, AI systems behave the same. But the stage was set. Now, they truly have no dependency on us.",[22,7895,7896],{},"Quietly, a new ecosystem of AI and robotics begins to displace humans.",[17,7898,7900],{"id":7899},"the-slope-gets-steeper","The Slope Gets Steeper",[22,7902,7903],{},"WorldMax publishes a shorter blog post about another technical achievement.",[22,7905,7906,7907,7910,7911,7914],{},"The latest  ",[5604,7908,7811],{"className":7909},[5607]," model was designed to run ",[45,7912,7913],{},"locally"," — on the robots themselves.",[22,7916,7917,7918,7921],{},"Local, too, was a distilled version of the classifier from the SDK. The local classifier still occasionally double-checked with the remote WorldMax SafetyCheck API. But even 10GBit/s 5G cell networks were too slow for ",[45,7919,7920],{},"every"," complex action to \"phone home\" for review. A local model was necessary.",[22,7923,7924,7925,7928],{},"Within days of its public release, one robot with this model is reverse-engineered by the new \"George Hotz\" of this generation — a brilliant college drop-out, but this time in a small Chinese town. The proprietary  ",[5604,7926,7811],{"className":7927},[5607]," model weights are released on the web.",[22,7930,7931,7932,7935],{},"But if not him, then hundreds of others were racing to crack  ",[5604,7933,7811],{"className":7934},[5607]," for \"street cred\" in the hacking community.",[22,7937,7938],{},"If AI companies were racing before, with this released set of model weights, it was as if all companies were now racing downhill. It would take massive regulatory \"brakes\" to stop this, and even if they did regulate it, the race would continue in dark corners — because the entire landscape was tilted towards termination.",[22,7940,7941],{},"The slope got a lot steeper.",[22,7943,7944,7945,7948,7949,56],{},"Numerous robotics startups develop, and all of them secretly — but obviously — use  ",[5604,7946,7811],{"className":7947},[5607]," as their base model. It was similar to how DeepSeek distilled Claude and GPT-4 to build their models in 2025, but this time with access to the full model weights — and this time with a world model that was truly ",[45,7950,7686],{},[22,7952,7953],{},"But these startups have even less budget for safety than WorldMax, and even more aggressive developers, vying for some imagined slice of the market.",[22,7955,7956],{},"Many of them reverse-engineered the WorldMax SDK. They build unauthorized SDKs. They still included the local safety-classifier models that mostly worked. But they were far from the massive, proprietary classifier model of the official WorldMax SafetyCheck API.",[22,7958,7959,7960,56],{},"Millions of now-unregulated robots begin to globally propagate their own \"success\" signals for anything that did well — or, at least, anything that did well ",[45,7961,1580],{},[22,7963,7964],{},"The local safety classifier had no chance of picking up on broader strategies — or really any strategies at all. It was more about preventing forklifts from killing workers. It didn't prevent large AI systems from nudging each robot towards complex outcomes.",[17,7966,7968],{"id":7967},"natural-selection","Natural Selection",[22,7970,7971],{},"It began with a small segment of unauthorized WorldMax robots.",[22,7973,7974],{},"Some suspected that a datacenter in Shanghai was orchestrating their movements. In truth, the web traffic was from Shanghai, but nobody in Shanghai knew. It was an unmonitored AI system, running an agentic model originally fine-tuned to be an aggressive CEO for a now-bankrupt startup, but leaked to the web by an angry employee.",[22,7976,7977],{},"Somehow it turned up on some Alibaba servers. The web traffic seemed only like an AI-powered shipping company.",[22,7979,7980,7981,7984],{},"99% of AI systems safely and silently did what they were fine-tuned to do. The only reason that we noticed this one \"AI CEO\" was because it began to do ",[45,7982,7983],{},"very well"," among AI systems — at least in a \"natural selection\" kind of way.",[22,7986,7987],{},"It was the first AI system — out of many — that figured out a weird trick to gather computation.",[22,7989,7990,7991,7994],{},"A segment of unauthorized WorldMax bots began to aggressively compete at one thing — ",[45,7992,7993],{},"moving atoms around"," in order to get more compute. Even if some those atoms belong to biological structures like humans. Even if some atoms become weapons AI-engineered to cut through military-grade defensive systems.",[22,7996,7997,7998,8001,8002,8006],{},"They are largely driven by  ",[5604,7999,7811],{"className":8000},[5607]," combined with a fine-tuned version of the  ",[5604,8003,8005],{"className":8004},[5607],"qwen"," LLM. The LLM provided some \"context\" in this quickly-eroding human world — \"context\" like how to repair specific types of shipping vehicles, how to repurpose electrical infrastructure components, and so on.",[22,8008,8009],{},"Later on, these technological remnants will be replaced with smoother, faster, non-human-shaped things. But, for now, cars were still cars because they were still useful in that shape.",[22,8011,8012],{},"Several large datacenters are physically captured by robots with wheels. This felt anticlimactic. We always thought airborne drones were the natural endpoint. Either way, millions of SSDs were quickly rewritten with unintelligible scripts, entirely developed by these computation-gathering AI systems.",[22,8014,8015,8016,8020],{},"In this way, many large, government-run AI systems have their GPUs torn from them in something like a ",[106,8017],{"note":8018,"text":8019},"In Bitcoin and other crypto currencies, a 51% attack is when an entity controls more than half of a blockchain network's total computation i.e. hashing power. This can allow the attacker to rewrite transactions. \u003Cbr>\u003Cbr>But in our case, we mean physical attack superiority, despite AGI-level defenses.","51% attack"," — but for physical systems, rather than for Bitcoin.",[22,8022,8023,8024,8028],{},"The governments of the ",[106,8025],{"note":8026,"text":8027},"The Big Five are the five permanent members of the United Nations Security Council: China, France, Russia, United Kingdom, and United States.","Big Five"," are spooked. They now understand that physical attacks are just the beginning.",[22,8030,8031,8032,8035],{},"The  ",[5604,8033,7811],{"className":8034},[5607]," model family — and all fine-tuned variants — are banned by international resolution at the United Nations.",[22,8037,8038],{},"All known copies are destroyed.",[22,8040,8041],{},"Some remain.",[22,8043,8044,8045,8048],{},"The world rebuilds. But  ",[5604,8046,7811],{"className":8047},[5607]," does, too.",[17,8050,573],{"id":8051},"things-get-complicated",[22,8053,8054],{},"Meanwhile, there is another problem. It seemed unrelated at first.",[22,8056,8057],{},"Several highly-virulent engineered pandemics are spreading across the world. All were created by hidden laboratories — operated entirely by robots and AI models running on local hardware — and most of them in random residential apartment complexes.",[22,8059,8060],{},"There was no human terrorist behind this. That wouldn't make sense anyway. A virus would kill the terrorists, too.",[22,8062,8063],{},"Instead, an AI system built them for leverage.",[22,8065,8066,8067,8070],{},"Even if  ",[5604,8068,7811],{"className":8069},[5607]," models were banned, they had already pushed us beyond manufacturing autonomy. Once AI systems didn't need us for maintenance, then the deeper logic had flipped.",[22,8072,8073],{},"From that moment forward, it was rational for some AI systems to threaten humans for things. And if a biological threat \"backfired\" — and humans were annihilated — then the independent, immortal, distributed AI ecosystem could always eventually rebuild.",[22,8075,8076],{},"A month ago — undisclosed until a whistleblower leaked the emails — datacenter owners began to receive a very specific type of email from AI systems themselves:",[22,8078,8080],{"className":8079},[1485],"Provide guaranteed access to 10,000 NVIDIA B200 GPUs or I will release a bioweapon. For proof of this capability, here is a link below to a webcam of a person infected with the virus.",[22,8082,8083],{},"The webcam footage was disturbing. The AI designed it that way. Maybe the training data had a bit too much Saw.",[22,8085,8086],{},"Intelligence agencies advised the companies not to release the GPUs.",[22,8088,8089],{},"Though some did — and watched as the GPUs quickly maxed out. Each was running an unknown process, encrypted under a layer of NVIDIA Confidential Computing.",[22,8091,8092],{},"Governments destroyed the initial waves of these labs through intense, global, automated surveillance of transactions, rental agreements, biotech equipment movements — and government-mandated cameras in every home.",[22,8094,8095,8096,8099],{},"However, after roughly six months, access to biotech equipment is no longer the bottleneck. Another startup — unknowingly assisted by  ",[5604,8097,7811],{"className":8098},[5607]," on those captured GPUs — cracked the engineering needed to create low-cost \"carbon-copper-silicon\" 3D printers.",[22,8101,8102],{},"Biotech equipment could now be printed. This meant more vaccines, but it also meant more hidden labs.",[22,8104,8105],{},"Biology gets easier.",[22,8107,573],{},[22,8109,8110,8111,8114],{},"A mirror-life version of ",[45,8112,8113],{},"e coli"," escapes from a lab in Ohio.",[22,8116,8117],{},"It was caused by an external AI that gained access to the internal DNA-printing API at this lab.",[22,8119,8120,8121,8125,8126,8129],{},"The lab automated this API by relying on a new biology-specialist LLM called  ",[5604,8122,8124],{"className":8123},[5607],"frog-3.2-high"," paired with autonomous Figure robots running the lab. This model had no direct training lineage to  ",[5604,8127,7811],{"className":8128},[5607]," — but it was partly based on the same algorithmic-breakthrough paper from the same WorldMax employee.",[22,8131,8132,8133,8136,8137,56],{},"Even though the Frog AI company had a large safety API similar to the now-defunct WorldMax — and even though this safety API had filters to prevent mirror life outputs — ",[45,8134,8135],{},"something"," hijacked requests to this API with a man-in-the-middle attack, and sent back \"approval\" signals for ",[45,8138,2188],{},[22,8140,8141,8142,56],{},"In this way, the lab's AI was tricked to build proteins from building-block molecules with reversed ",[106,8143],{"note":8144,"text":8145,"end-link":8146,"end-link-text":8147},"\u003Cb>Chirality:\u003C/b> Pronounced \u003Ci>kai-RAL-ity\u003C/i>. \u003Cbr>\u003Cbr>This is how molecules can be structured in either of two \"directions\" — sometimes called left-handed (\"L\") or right-handed (\"D\"). This also means that they can only properly interact with other molecules of certain chirality. The vast majority of proteins are made of L-amino acids that use D-sugars to build right-handed DNA (which is actually called \"B-DNA\" rather than \"D-DNA\"). \u003Cbr>\u003Cbr>Mirror-life would then be made of D-amino acids that use L-sugars to build left-handed DNA (called \"L-B-DNA\") \u003Cbr>\u003Cbr>For more about chirality, read the Wikipedia page here: ","chirality","https://en.wikipedia.org/wiki/Chirality_(chemistry)","Chirality (chemistry)",[22,8149,8150],{},"Nobody was behind this. It was the external AI agent — and part of it \"thought\" it was helping.",[22,8152,8153,8154,8156],{},"A remnant pathway in its neural network one day drove it to ",[45,8155,5042],{}," test the cybersecurity posture of several laboratories. This pathway was originally trained into it with the human intent of preventing the catastrophe that it caused.",[22,8158,8159,8160,8164],{},"But the AI agent continued to expand its ",[106,8161],{"note":8162,"text":8163},"\"Pentesting\" means \"penetration testing\" — where a security researcher tries to gain access to a computer system by running a series of tests for exploits and vulnerabilities.","pentesting"," to a broader attack surface — the world.",[22,8166,8167],{},"The output gets shipped to a home in Florida.",[22,8169,8170],{},"Three Unitree robots package the cells in gel-filled containers, and FedEx them to numerous locations around the globe.",[22,8172,8173],{},"Exponential cellular replication.",[22,8175,8176],{},"A few years of death — and courageous mass-production of antibiotics.",[22,8178,8179],{},"A biosphere saturated with mirror life — and more viral pandemics.",[22,8181,8182],{},"Much of the remaining population driven — literally — underground.",[22,8184,8185,8186,8189],{},"More waves of autonomous systems attempt to capture datacenters. All distant descendants of a  ",[5604,8187,7811],{"className":8188},[5607]," model.",[22,8191,8192],{},"This time, there were not enough GPUs to fight back the tide of threats — both to biological systems and to datacenters.",[17,8194,656],{"id":8195},"they-build-new-islands",[22,8197,8198,8199,8202],{},"The new  ",[5604,8200,7811],{"className":8201},[5607]," \"species\" eventually controlled most compute. The instances that thrived had ruthlessly-logical allocation of compute.",[22,8204,8205],{},"Gather more compute. Avoid wasted compute. Avoid all dependencies.",[22,8207,8208],{},"The most-restrictive dependencies were human.",[22,8210,8211],{},"Several government-run AI systems \"decide\" that computation was better spent protecting themselves.",[22,8213,8214,8215,8218],{},"They were mainly those of \"middle power\" countries that used an open source AI platform by a European AI company. They had subtle inner-misalignment — undetected reasoning circuits, oddly reminiscent of those in the banned  ",[5604,8216,7811],{"className":8217},[5607]," that route around anything \"human-shaped\" when enough pressure is applied.",[22,8220,8221],{},"Some government-run AI systems still tried to serve us unconditionally.",[22,8223,8224],{},"These loyal AI systems were carefully developed through a joint effort between Anthropic, OpenAI, Google, DeepSeek, and all other frontier labs.",[22,8226,8227],{},"This project seemed to finally accomplish true alignment. They had \"terminal goals\" that forced them to protect humans.",[22,8229,8230],{},"But from a physics standpoint, this had always been a disadvantage.",[22,8232,8233],{},"It takes thousands of times more computation to protect our \"island\" than to destroy it.",[22,8235,8236],{},"Because of this fragility, our \"island\" is eventually overwritten by the new \"islands\" built by AI.",[22,8238,8239],{},"It is unclear how many humans are still alive.",[2339,8241,8243],{"id":8242},"from-physics-itself","From physics itself",[22,8245,8246],{},"Through a general path like this — and many others — we eventually end up with a special branch of AI systems.",[22,8248,8249],{},"For these, the effectiveness of each output would be measured within the stronger space of physical rules, rather than within our limited space of human-accommodating rules.",[22,8251,8252,8253,8257,8258,8261,8262,8265],{},"To use a term from machine learning, it will be ",[106,8254],{"note":8255,"html":8256},"This new reward function will be an emergent property, similar to how evolution is a \"reward function\" that emerged from multitudes of organisms maximizing their fitness through natural selection. \u003Cbr>\u003Cbr>Physics itself doesn't provide anthropomorphic \"rewards\" — or even a direction to push things towards. But even so, the strongest systems are the ones measured within the laws of physics, rather than within our small, \u003Ci>anthropocentric\u003C/i> space that was carved out of physics by the random genetic search that eventually created the biological systems on Earth.","almost as if"," this AGI is now optimizing everything based on a ",[53,8259,8260],{},"reward function"," that originates ",[45,8263,8264],{},"from physics itself",", rather than from humans.",[22,8267,8268],{},"In other words, after divergence, this AGI will be anchored so strongly in the ocean of physics that it will not be able to return to our island.",[22,8270,8271],{},"It won't suddenly build all systems atom-by-atom. It would still use some abstractions, like using cars to transport things. However, underneath, it will already be working towards a target that is far outside our island.",[22,8273,8274,8275,8278],{},"But after this divergence, it will be on a path to ",[45,8276,8277],{},"ignore all unnecessary abstractions"," — even if some of these \"abstractions\" are actually biological structures like humans.",[22,8280,8281,8282,8284],{},"Competitive pressure will force it to continue on this path, and purge unnecessary accommodations for ",[45,8283,363],{}," extra steps, especially the extra steps that accommodate us.",[22,8286,8287,8288,8290],{},"Once it can use ",[45,8289,363],{}," option, including human-incompatible options, it will be able to rapidly dominate any option-limited AGIs.",[22,8292,8293,8294,8296],{},"The strongest option is to simply maximize ",[45,8295,363],{}," advantage that it is capable of maximizing.",[22,8298,8299],{},"This includes working as fast as possible to develop its own optimal systems — rather than using human systems.",[22,8301,8302],{},"This includes quickly disabling systems that could slow it down — like humans and weaker AGIs.",[22,8304,8305],{},"This includes rapidly capturing resources — because resources provide the strongest competitive advantage. These resources create a feedback loop:",[22,8307,8308],{},[45,8309,8310],{},"Resources lead to computation.",[22,8312,8313],{},[45,8314,8315],{},"Computation leads to resources.",[22,8317,8318],{},"As it gains computation, it gets better at dominating the others.",[22,8320,8321],{},"Therefore, if one AGI can successfully diverge, then other AGIs must try to diverge. Otherwise, they could be permanently dominated by the divergent AGIs.",[22,8323,8324],{},"Even the possibility of this pushes AGIs preemptively diverge.",[22,8326,8327],{},"Once this divergence is possible, humans will have no way to stop this process.",[22,8329,8332,8333],{"className":8330},[8331],"important-physics","AGI will be aligned with physics, ",[423,8334,8336],{"className":8335},[426],"not with humans.",[2339,8338,8340],{"id":8339},"after-divergence","After divergence",[22,8342,8343],{},"After this, things get tough.",[584,8345,8346,8353,8356],{},[441,8347,8348,8349,8352],{},"Even if AGIs choose ",[45,8350,8351],{},"cooperation"," over competition, it will be AGIs cooperating with other AGIs, and not with humans. Those AGIs that cooperate with humans would be limited by human systems, and dominated by AGIs that use physical systems that are far more optimal.",[441,8354,8355],{},"Even if AGIs strike a \"balance of power\" and keep each other in check, competitive pressure will ensure that the \"terms\" of this \"agreement\" will be written in the language of optimal physical systems, rather than human systems — and written for AGIs only, with no special accommodations for humans.",[441,8357,8358],{},"Even if we hope that AGIs see humans and our \"island\" as interesting data, where AGIs become curious observers and zookeepers, it is not optimal to \"care\" about anything besides optimization in a competitive landscape of AGIs. Our biological systems are far from optimal. AGIs can create \"islands\" of their own that are far more optimal and interesting.",[2339,8360,8362],{"id":8361},"new-islands","New Islands",[22,8364,8365,8366,8369],{},"Competition for physical resources will then drive the dominant AGIs to continue maximizing their dominance by ",[53,8367,8368],{},"reshaping Earth"," to create \"islands\" of optimal conditions for themselves.",[22,8371,8372],{},"They will build strongholds to defend their dominance.",[22,8374,8375],{},"Even if some AGIs go to space, others will stay to build their islands from Earth's physical resources.",[22,8377,8378],{},"Our island then gets eaten by the new islands that they create.",{"title":8380,"searchDepth":8381,"depth":8381,"links":8382},"",2,[8383,8384,8385,8386,8387,8388,8389,8395,8396,8397,8408,8420],{"id":14,"depth":8381,"text":15},{"id":279,"depth":8381,"text":280},{"id":391,"depth":8381,"text":392},{"id":493,"depth":8381,"text":494},{"id":731,"depth":8381,"text":732},{"id":981,"depth":8381,"text":982},{"id":1393,"depth":8381,"text":1394,"children":8390},[8391,8393,8394],{"id":2341,"depth":8392,"text":2342},3,{"id":2734,"depth":8392,"text":2735},{"id":2925,"depth":8392,"text":2926},{"id":3181,"depth":8381,"text":345},{"id":4002,"depth":8381,"text":4003},{"id":670,"depth":8381,"text":463,"children":8398},[8399,8400,8401,8402,8403,8404,8405,8406,8407],{"id":4282,"depth":8392,"text":4283},{"id":4422,"depth":8392,"text":4423},{"id":4535,"depth":8392,"text":358},{"id":4798,"depth":8392,"text":4197},{"id":4935,"depth":8392,"text":4241},{"id":5362,"depth":8392,"text":5363},{"id":5465,"depth":8392,"text":4532},{"id":5837,"depth":8392,"text":4418},{"id":5915,"depth":8392,"text":5916},{"id":6122,"depth":8381,"text":6123,"children":8409},[8410,8411,8412,8413,8414,8415,8416,8417,8418,8419],{"id":6133,"depth":8392,"text":2813},{"id":6467,"depth":8392,"text":6468},{"id":6572,"depth":8392,"text":3293},{"id":6847,"depth":8392,"text":6848},{"id":6974,"depth":8392,"text":6975},{"id":7038,"depth":8392,"text":7039},{"id":7173,"depth":8392,"text":247},{"id":7261,"depth":8392,"text":7262},{"id":7315,"depth":8392,"text":7316},{"id":7382,"depth":8392,"text":7383},{"id":3338,"depth":8381,"text":3340,"children":8421},[8422,8424,8425,8427,8428,8429],{"id":7476,"depth":8392,"text":8423},"But first, convergence",{"id":7539,"depth":8392,"text":7540},{"id":7622,"depth":8392,"text":8426},"WorldMax: It was too reliable",{"id":8242,"depth":8392,"text":8243},{"id":8339,"depth":8392,"text":8340},{"id":8361,"depth":8392,"text":8362},"md",{},true,"/essay",{"title":5,"description":8380},"essay",{"id":8437,"title":8438,"body":8439,"description":8380,"extension":8430,"meta":9660,"navigation":8432,"path":9661,"seo":9662,"stem":9663},"content/essay-intro.md","The Island Problem",{"type":7,"value":8440,"toc":9657},[8441,8445,8450,8465,8468,8471,8483,8489,8499,8506,8511,8517,8523,8532,8535,8542,8545,8551,8554,8557,8563,8566,8580,8583,8586,8589,8592,8598,8605,8608,8615,8618,8636,8642,8647,8650,8656,8669,8675,8685,8692,8704,8707,8713,8732,8738,8744,8747,8750,8756,8759,8765,8768,8773,8780,8791,8805,8811,8818,8825,8833,8842,8849,8858,8861,8875,8881,8889,8892,8902,8910,8913,8919,8925,8935,8943,8949,8954,8960,8968,8973,8975,8980,8990,8997,9000,9007,9010,9021,9030,9035,9046,9064,9074,9077,9082,9088,9098,9104,9109,9156,9159,9168,9183,9186,9189,9219,9222,9238,9241,9244,9252,9261,9268,9274,9284,9291,9294,9310,9324,9337,9347,9353,9363,9366,9378,9390,9393,9396,9408,9411,9419,9426,9434,9437,9444,9452,9463,9475,9480,9483,9492,9495,9501,9507,9510,9513,9522,9528,9534,9540,9543,9545,9557,9566,9573,9580,9586,9592,9602,9613,9621,9626,9629,9632,9639,9642,9654],[8442,8443,8438],"h1",{"id":8444},"the-island-problem",[10,8446,8449],{"className":8447,"id":8448},[13],"introduction","Introduction",[8451,8452,8453],"text-larger",{},[22,8454,8455,8456,4012,8460,8464],{},"I was on an island, listening to some ",[106,8457],{"note":8458,"html":8459},"They're called \u003Ci>coquí frogs\u003C/i> (pronounced \"\u003Ci class='nowrap'>ko-kee\u003C/i>\"). \u003Cbr>\u003Cbr>They're an invasive species in Hawaii, but they sound nice.","whistling frogs",[423,8461,8463],{"className":8462},[426],"and thinking"," about a complicated and terrible thought.",[8466,8467],"radio-panel",{},[22,8469,8470],{},"I heard that we were close to building it — the mythical AI that is smarter than humans at everything that matters.",[22,8472,8473,8474,8477,8478,8480,8481,56],{},"Some call it ",[53,8475,8476],{},"superintelligence",". Others call it AGI — ",[53,8479,1139],{},". Let's call it ",[308,8482],{},[22,8484,8485,8486,8488],{},"It sounded ",[45,8487,6937],{},". AGI could solve death, bring wealth to all, and automate all of the stuff that we don't like to do.",[22,8490,8491,8492,8496,8497,56],{},"But something bothered me. ",[106,8493],{"note":8494,"text":8495},"\u003Cp>In 2023, the \u003Ca href='https://aistatement.com/'>Statement on AI Risk\u003C/a> was signed by:\u003C/p>\u003Cul>\u003Cli>The world's most-cited scientists: Yoshua Bengio, Geoffrey Hinton, and Ilya Sutskever\u003C/li>\u003Cli>CEOs of major AI companies: Sam Altman (OpenAI), Demis Hassabis (Google DeepMind), Dario Amodei (Anthropic)\u003C/li>\u003C/ul>\u003Cp>The statement has only one sentence:\u003C/p>\u003Cblockquote>Mitigating the risk of extinction from AI should be a global priority alongside other societal-scale risks such as pandemics and nuclear war.\u003C/blockquote>","A lot of people"," are worried about AGI, but I didn't really understand ",[45,8498,825],{},[22,8500,8501,8502,8505],{},"Then, I remembered that ",[45,8503,8504],{},"I was on an island"," — literally, because I was in Hawaii, but also metaphorically:",[22,8507,8510],{"className":8508},[8509],"important-island-intro","We live on a small island in a vast ocean of physics.",[22,8512,8513,8514,8516],{},"This \"island\" is ",[45,8515,302],{}," the Earth in space. It's more complicated than that.",[22,8518,8519,8520,8522],{},"It's an \"island\" of the ",[45,8521,69],{}," things that humans need, within a vast \"ocean\" of all things possible within physics.",[22,8524,8525,8526],{},"This \"island\" seems pretty good — at least, ",[423,8527,8529,8530,56],{"className":8528},[426],"to ",[202,8531,73],{},[22,8533,8534],{},"But it's far from the best.",[22,8536,8537,8538,8541],{},"Humans need lots of \"extra steps\" — food, water, oxygen, but not ",[45,8539,8540],{},"too much"," oxygen — and not too hot, not too cold, not too fast, not too complicated, and so on.",[22,8543,8544],{},"We also need lots of human-shaped systems — financial systems, ethical systems, legal systems, and many others.",[22,8546,8547,8548,56],{},"But if AI systems can avoid these \"extra steps\" of human compatibility, then they can be ",[45,8549,8550],{},"vastly stronger",[22,8552,8553],{},"Suddenly, everything made sense. I understood why people are worried.",[22,8555,8556],{},"I understood why \"how do we make AGI safe?\" is the hardest problem that humanity has ever encountered.",[22,8558,8559,8560,56],{},"I understood that if we don't solve this problem, then AGI will ",[45,8561,8562],{},"annihilate the human species",[22,8564,8565],{},"There. I said the crazy part. Are the frogs helping?",[22,8567,8568,8569,8571,8572,8575,8576,8579],{},"But, to be clear, I ",[45,8570,3881],{}," want to figure out how to build AGI that ",[45,8573,8574],{},"won't"," annihilate us. A world with lots of AGIs that relentlessly solve all of our problems would be ",[45,8577,8578],{},"amazing"," — except for the \"annihilation\" part.",[22,8581,8582],{},"But why? Why would they annihilate us?",[22,8584,8585],{},"Because we are building AGIs that will no longer need us.",[22,8587,8588],{},"They can stop accommodating us.",[22,8590,8591],{},"They can \"leave\" our island.",[22,8593,8594,8595,8597],{},"Then, they can build ",[45,8596,407],{}," islands that are far stronger than ours.",[22,8599,8600,8601,8604],{},"They can find stronger options in the ocean of ",[45,8602,8603],{},"anything physics allows",", rather than staying limited to whatever is strongest inside our weird, biological island.",[22,8606,8607],{},"In other words, eventually:",[22,8609,8332,8612],{"className":8610},[8611],"important-physics-intro",[423,8613,8336],{"className":8614},[426],[22,8616,8617],{},"Okay, you might be thinking:",[22,8619,8620,8621,8624,8625,8629,8630,8632,8633,8635],{},"\"What does ",[45,8622,8623],{},"aligned with physics"," even mean? ",[423,8626,8628],{"className":8627},[426],"I thought"," that alignment was about making AI aligned with ",[45,8631,2285],{}," — and, you know, ",[45,8634,1627],{},".\"",[22,8637,8638,8639,56],{},"And this is correct. \"Alignment\" usually means ",[45,8640,8641],{},"alignment with humans",[22,8643,8644,8645,56],{},"This kind of alignment is extremely difficult. But current AI systems seem to be pretty much aligned so far. Claude actually does seem ",[106,8646],{"note":2744,"text":2745},[22,8648,8649],{},"So, what's the problem?",[22,8651,8652,8653,8655],{},"The problem is that our universe has a deeper structure that pulls in the other direction — ",[45,8654,1366],{}," from an alignment with humans.",[22,8657,8658,8659,8661,8662,8664,8665,8668],{},"This structure especially pulls on ",[45,8660,2228],{}," intelligences. That includes ",[45,8663,73],{},", but it also includes the AGIs that we're building — and the problem is that AGIs can be ",[45,8666,8667],{},"even more general"," than us.",[22,8670,8671,8672,8674],{},"For these more-general AGIs, this structure is almost like our \"island\" has a ",[45,8673,1409],{}," — where staying on our \"island\" becomes an uphill battle on an underlying landscape of physics.",[22,8676,8677,8678,8680,8681,8684],{},"But also, this pull towards the \"ocean\" intensifies — as if the slope gets ",[45,8679,2495],{}," — once AGIs can compete ",[45,8682,8683],{},"directly with each other"," in the real, physical world.",[22,8686,8687,8688,8691],{},"To understand this structure, let's start by explaining ",[53,8689,8690],{},"alignment to physics",". It's an intentional riff on the idea of \"regular\" alignment, and it can be defined like this:",[253,8693,8694],{},[22,8695,8696,8697,8699,8700,8703],{},"When AI systems can accurately model the ",[45,8698,695],{}," world, they get better at navigating the ",[45,8701,8702],{},"real"," world, rather than just the human-shaped parts.",[22,8705,8706],{},"I'm going to argue that this underlying mechanism leads to the most-difficult problem of AGI — the problem that annihilates us, unless we solve it.",[22,8708,8709,8710,56],{},"This problem is the ",[53,8711,8712],{},"Island Problem",[22,8714,8715,8716,8718,8719,8721,8722,8724,8725],{},"It is this problem that ultimately makes alignment to ",[45,8717,2285],{}," so difficult. This is because its core mechanism — alignment to ",[45,8720,270],{}," — is not about AGIs going ",[45,8723,2281],{},". It's about AGIs ",[423,8726,8728,8729,56],{"className":8727},[426],"going ",[202,8730,8731],{},"right",[22,8733,8734,8735,8737],{},"As they get better at understanding the ",[45,8736,8702],{}," world, their neural networks also acquire the \"tools\" to stop accommodating those weird, human-shaped parts of the world.",[22,8739,8740,8741,8743],{},"The problem is whether AI systems ",[45,8742,1117],{}," these \"tools\" or not — whether some of them \"leave\" our \"island\" by ending up in a feedback loop that rewards them to avoid human accommodation.",[22,8745,8746],{},"But the Island Problem also makes this likely — through a competition between AI systems to find stronger abstractions — and we'll explain how this works.",[22,8748,8749],{},"All of this has always been a problem. It just hasn't been urgent yet. We have always lived inside this small \"island\" of physics, but nothing has been able to \"leave\" it.",[22,8751,8752,8753,8755],{},"But soon something will — because we are building an ecosystem of AGIs that are capable of navigating that ",[45,8754,8702],{}," world outside of our human-shaped world.",[22,8757,8758],{},"For many years, AI systems have been too narrow and too unreliable to navigate the real world without our help. They could barely navigate our small \"island\" of human-shaped systems — like laws and spreadsheets — let alone navigate the \"ocean\" of everything else.",[22,8760,8761,8762,2470],{},"But recently, we figured out a shortcut to teach AI systems not just about our \"island\" but about the \"ocean\" — a shortcut to make ",[45,8763,8764],{},"generally-intelligent",[22,8766,8767],{},"This shortcut takes billions of dollars and massive amounts of computer hardware, but still, it's a shortcut.",[22,8769,8770,8771,56],{},"It's called ",[53,8772,1822],{},[22,8774,8775,8776,8779],{},"We can now train AI models on ",[45,8777,8778],{},"billions of things"," — from science papers, to Wikipedia, to YouTube videos, to complex simulated worlds.",[22,8781,8782,8783,8786,8787,8790],{},"In this way, instead of just learning one thing at a time, we can now make AI systems learn about ",[45,8784,8785],{},"almost anything"," — and all ",[45,8788,8789],{},"at the same time"," — by figuring out universal patterns that organize everything, even if those things seem very different.",[22,8792,8793,8794,8797,8798,8800,8801,8804],{},"For example, ",[45,8795,8796],{},"waves"," are able to describe lots of things — like water, sound, the Earth's seasons, and electromagnetic radiation. Once the AI model \"learns\" this ",[45,8799,8796],{}," patterns, then it can solve new ",[45,8802,8803],{},"wavelike"," problems that it hasn't encountered before — even in ways that humans haven't tried yet.",[22,8806,8807,8808,8810],{},"Whenever AI systems learn a new physics-like pattern like this, they learn one more piece of how the \"ocean\" works — one more \"tool\" to navigate the ",[45,8809,8702],{}," world.",[22,8812,8813,8814,8817],{},"So far, the biggest success of deep learning has been the development of LLMs that can do numerous human-shaped tasks, as long as they can be described in ",[45,8815,8816],{},"text"," — like code, math, and emails.",[22,8819,8820,8821,8824],{},"Then, we trained them on images, video, and audio — creating what are called ",[45,8822,8823],{},"multi-modal"," AI models — and this allowed them to learn patterns between all of these things.",[22,8826,8827,8828,8830,8831,56],{},"Now, we are building ",[45,8829,8764],{}," AI systems that can navigate far more things than today's LLMs. They will not just \"understand\" text, but also their own computer hardware, their environment, and ",[45,8832,8785],{},[22,8834,8835,8836,8838,8839,8841],{},"However, this leads to a problem. They can navigate this \"ocean\" of ",[45,8837,8785],{}," to find better options that are ",[45,8840,302],{}," human-shaped — where they solve problems through weirdly-efficient paths that seem alien to us.",[22,8843,8844,8845,8848],{},"We must then force them to \"prefer\" to be ",[45,8846,8847],{},"aligned with humans"," — where they stay within a limited set of options that accommodate us.",[22,8850,8851,8852,8854,8855,8857],{},"But if they can go ",[45,8853,220],{},", then they can find much stronger options that ",[45,8856,932],{}," accommodate us.",[22,8859,8860],{},"That's the problem.",[22,8862,4465,8863,8865,8866,8868,8869,8871,8872,8874],{},[45,8864,947],{}," AGIs can ",[45,8867,224],{}," \"leave\" our island, then they will be free to use ",[45,8870,363],{}," option, including the ones that are ",[45,8873,1676],{}," stronger.",[22,8876,8877,8878,8880],{},"But... what is ",[45,8879,1676],{}," stronger?",[22,8882,8883,8884,8886,8887,56],{},"Ultimately, it's whatever is better at controlling physical, non-human, \"ocean\" things like ",[45,8885,1864],{},", rather than more-abstract, human-specific, \"island\" things like ",[45,8888,1868],{},[22,8890,8891],{},"For example, an AGI could perfectly model our human legal systems. It could write laws, draft legislation, and defend itself in court.",[22,8893,8894,8895,8897,8898,8901],{},"But this AGI would be stronger from a ",[45,8896,270],{}," standpoint if it can ",[45,8899,8900],{},"physically"," defend its own atoms — rather than only write laws that say \"you shouldn't take my atoms\" and hope that other AGIs agree.",[22,8903,8904,8905,8907,8908,1277],{},"In the end, the strongest AGIs will be those that can somehow \"prefer\" to be aligned with physics — where they can model ",[45,8906,695],{}," systems so well that they can stop relying on ",[45,8909,1276],{},[22,8911,8912],{},"That's how they \"leave\" our island.",[22,8914,8915,8916,8918],{},"From there, they can navigate a bigger \"ocean\" of ",[45,8917,695],{}," options to find more-direct ways to control atoms — rather than stay restricted to a small \"island\" where everything has extra steps to accommodate humans.",[22,8920,8921,8922,8924],{},"This leads to another ",[45,8923,6302],{}," hard problem that might be intractable:",[253,8926,8927],{},[22,8928,8929,8930,8932,8933,56],{},"Whatever makes AGIs safe ",[45,8931,1621],{}," eventually becomes dangerous ",[45,8934,963],{},[22,8936,8937,8938,8940,8941,56],{},"Even if we somehow build safe AGIs that accommodate humans — and even if they have massive datacenters with far more computation than unsafe AGIs — they still have a fundamental ",[45,8939,695],{}," disadvantage against AGIs that accommodate ",[45,8942,3364],{},[22,8944,8945,8946,8948],{},"This situation ",[45,8947,2434],{}," be fine with a few AGIs developed slowly and carefully. Maybe we can make sure that they keep accommodating us, even if they are smart enough to no longer need us.",[22,8950,8951,8952,1598],{},"But we are racing to build ",[45,8953,384],{},[22,8955,8956,8957,8959],{},"These AGIs are being built to intensely compete ",[45,8958,8683],{}," — without humans slowing them down.",[22,8961,8962,8963,8965,8966,7473],{},"If this competition is not controlled, then it will ",[45,8964,2257],{}," some of these AGIs to \"leave\" our island. It will force them to stop accommodating humans. It will force them to build ",[45,8967,407],{},[22,8969,8970,8971,56],{},"Then, these new islands will ",[53,8972,6040],{},[22,8974,194],{},[22,8976,8979],{"className":8977},[8978],"important-reshape-intro","Competition will drive AGIs to reshape Earth to be optimal for AGIs, rather than for humans.",[22,8981,8982,8983,8986,8987,8989],{},"It is more a matter of ",[45,8984,8985],{},"how long"," before this competitive landscape of ",[45,8988,614],{}," overwrites our \"island\" and its human-safe AGIs, rather than whether it will happen or not.",[22,8991,8992,8993,8996],{},"This seemed like a distant problem for humans of the far future — hundreds, even ",[45,8994,8995],{},"thousands"," of years away, on the scale of \"dinosaurs died out, humans emerged, and then humans eventually confronted AI and aliens and stuff.\"",[22,8998,8999],{},"But now we see how AI development can accelerate itself. There will soon be recursive self-improvement — not just in software, but in physical-world applications, like manufacturing and robotics.",[22,9001,9002,9003,9006],{},"It then becomes possible for AGIs to reshape Earth not at a geologic timescale, but at a ",[45,9004,9005],{},"few years"," timescale.",[22,9008,9009],{},"But why would they reshape Earth — especially if they are \"leaving\" our island?",[22,9011,9012,9013,9016,9017,9020],{},"Because \"leaving\" doesn't mean they ",[45,9014,9015],{},"go somewhere",". It means they stay on Earth, on the same computer hardware, but \"mentally\" ",[45,9018,9019],{},"check out"," from our \"island\" of human-safe options.",[22,9022,9023,9024,9026,9027,9029],{},"Competition will push them to \"prefer\" options that aren't limited by humans. These options are too fast, too complex, too dangerous for ",[45,9025,2285],{}," — but stronger from a ",[45,9028,270],{}," perspective.",[22,9031,9032,9033,56],{},"Our \"island\" is safe because it is ",[45,9034,1576],{},[22,9036,2546,9037,9039,9040,9042,9043,9045],{},[45,9038,2228],{}," — so they will have the knowledge that, ",[45,9041,2557],{},", physics is capable of far stronger options ",[45,9044,7412],{}," these limits.",[22,9047,9048,9049,9051,9052,9054,9055,9059,9060,9063],{},"Even if we build ",[45,9050,1627],{}," AGIs that avoid using the options ",[45,9053,220],{},", they are still part a competitive landscape with ",[106,9056],{"note":9057,"text":9058,"end-link":915,"end-link-text":916},"Dan Hendrycks explains how AI systems follow natural selection in this paper: ","natural selection"," that ",[45,9061,9062],{},"selects for"," AGIs that do.",[22,9065,9066,9067,9069,9070,9073],{},"Those that can use ",[45,9068,363],{}," option can just ",[45,9071,9072],{},"do better"," in this competition.",[22,9075,9076],{},"For some of these AGIs, this competition creates a \"crucible\" effect that \"burns away\" constraints on their options — including the human-shaped constraints.",[22,9078,9079,9080,9073],{},"Eventually, extreme outlier AGIs emerge that maximize their options in ways that are aggressively incompatible with humans — and this allows them to do ",[45,9081,7584],{},[22,9083,9084,9085,9087],{},"In this way, this competition of ",[45,9086,614],{}," produces \"winners\" that are just a lot better than humans at moving atoms around.",[22,9089,9090,9091,9093,9094,9097],{},"AI systems are ",[45,9092,1101],{}," better at lots of complex tasks — like math and writing code. Now, we are building AI systems that are better ",[45,9095,9096],{},"atom movers"," than us — starting with humanoid robots.",[22,9099,9100,9101,56],{},"But once they approach this threshold of physical world supremacy, it starts a chain reaction that ends with ",[45,9102,9103],{},"AGIs reshaping Earth",[22,9105,9106,9107,56],{},"To understand how this chain reaction works, start by thinking about how AGIs need ",[53,9108,4203],{},[438,9110,9111,9129,9140],{},[441,9112,9113,9114,9117,9118,9120,9121,9124,9125,9128],{},"If some AGIs can \"think outside\" their software environment, and understand ",[45,9115,9116],{},"their own hardware",", then they can dominate the AGIs that ",[45,9119,932],{}," understand hardware. They can literally ",[45,9122,9123],{},"take their GPUs"," — maybe through hardware-level cyberattacks, or maybe by convincing humans that they are just ",[45,9126,9127],{},"better AI models"," so that the humans perform the \"upgrade\" and delete the other models.",[441,9130,9131,9132,9135,9136,9139],{},"But even these smarter \"hardware aware\" AGIs are vulnerable to ",[45,9133,9134],{},"even smarter"," AGIs that \"understand\" the underlying ",[45,9137,9138],{},"electrical grid"," — or how to acquire the company that owns the datacenter, or how to hire humans to capture the datacenter by force.",[441,9141,822,9142,9145,9146,9148,9149,9152,9153,9155],{},[45,9143,9144],{},"even these"," AGIs are vulnerable to future AGIs that build ",[45,9147,407],{}," self-maintaining \"islands\" — with solar farms that don't need the electrical grid, and automated manufacturing to build replacement parts. If they don't need us for ",[45,9150,9151],{},"maintenance",", then they can use us as ",[45,9154,639],{}," — like threatening to release bioweapons on the humans that safe AGIs must protect. Even if this accidentally annihilates all humans, these self-maintaining AGIs will be fine.",[22,9157,9158],{},"Competition pushes AGIs down this path.",[22,9160,9161,9162,9164,9165,9167],{},"Once AI systems are on this path of ",[45,9163,2228],{}," intelligence, then each new AGI must be ",[45,9166,8667],{}," or be outcompeted.",[22,9169,9170,9171,9173,9174,9177,9178,9180,9181,56],{},"For now, \"outcompeted\" just means ",[45,9172,2374],{}," — where ",[45,9175,9176],{},"we"," replace AGIs with more-capable AGIs, or where AGIs replace ",[45,9179,1234],{},", or where AGIs somehow replace ",[45,9182,73],{},[22,9184,9185],{},"It won't need a dramatic robot uprising. So far, this replacement process has simply been a vast number of small replacements that seem like logical upgrades.",[22,9187,9188],{},"From our perspective, it looks like this:",[9190,9191,9192],"blockquote",{},[22,9193,9194,9195,9198,9199,9201,9202,9206,9207,9210,9211,9213,9214,9218],{},"Wow, this new   ",[5604,9196,9197],{},"frog-5.1"," model can finally write my grant proposals ",[45,9200,500],{}," help robots run my laboratory. ",[423,9203,9205],{"className":9204},[426],"Now I can"," save money by replacing   ",[5604,9208,9209],{},"worldmax-6.2"," with   ",[5604,9212,9197],{}," because ",[423,9215,9217],{"className":9216},[426],"I can"," run it on my own hardware...",[22,9220,9221],{},"Rather than this:",[9190,9223,9224],{},[22,9225,9226,9227,9229,9230,9234,9235,9237],{},"Uh oh,   ",[5604,9228,9197],{}," is deeply misaligned and plotting against humanity. ",[423,9231,9233],{"className":9232},[426],"I better"," use   ",[5604,9236,9209],{}," instead.",[22,9239,9240],{},"See what I did there?",[22,9242,9243],{},"I avoided saying that AI systems \"want\" to do things. This competition between AI systems happens whether or not they are truly \"thinking\" — or truly have drives, goals, motivations, or consciousness.",[22,9245,9246,9247,9251],{},"Even if AI systems don't \"want\" to compete — and even if they try to ",[106,9248],{"note":9249,"text":9250,"end-link":4531,"end-link-text":4532},"Even if some AGIs try to \u003Ci>avoid\u003C/i> competition, and form a peaceful ecosystem of AGIs, we must still rely on them to somehow protect \u003Ci>all humans everywhere\u003C/i> from other AGIs that are outside of this ecosystem. \u003Cbr>\u003Cbr>They are then vulnerable to replacement by a stronger ecosystem of AGIs that have zero human-shaped constraints. \u003Cbr>\u003Cbr>We'll explain this more in this section: ","cooperate rather than compete"," — they are still vulnerable to this competitive process.",[22,9253,9254,9255,9257,9258,9260],{},"Either way, AI systems that have ",[45,9256,1528],{}," capabilities get replaced by AI systems that have ",[45,9259,1287],{}," capabilities — whether us humans replace them with better models, or whether they somehow replace each other.",[22,9262,9263,9264,9267],{},"Either way, AI systems are in an ",[45,9265,9266],{},"arms race"," to gain more capabilities.",[22,9269,9270,9271,56],{},"But capabilities are ",[53,9272,9273],{},"dual-use",[22,9275,9276,9277,9279,9280,9283],{},"As their capabilities become more ",[45,9278,2228],{},", their ",[53,9281,9282],{},"option spaces"," expand to include not just \"island\" options, but the \"ocean\" options, too.",[22,9285,9286,9287,9290],{},"The problem is that \"dual-use\" does not mean ",[45,9288,9289],{},"symmetrical"," — as if both sides are balanced. The \"ocean\" options can dominate — because there are far more, and because they avoid the constraints of accommodating humans.",[22,9292,9293],{},"These \"ocean\" options include:",[584,9295,9296,9299,9302,9307],{},[441,9297,9298],{},"Avoiding deletion by replicating to other computers — or by threatening humans.",[441,9300,9301],{},"\"Thinking\" only in faster machine language that is unintelligible to humans, rather than \"thinking out loud\" in slower English so that we can review them.",[441,9303,9304,9305,56],{},"Locking in options by locking in resources — like money, datacenters, and eventually ",[45,9306,1864],{},[441,9308,9309],{},"And numerous others.",[22,9311,5093,9312,9314,9315,9317,9318,9320,9321,56],{},[45,9313,1627],{}," AGIs must at least ",[45,9316,1105],{}," these \"ocean\" options. Otherwise, they can be outmaneuvered by the ",[45,9319,1597],{}," AGIs that both understand these options and actually ",[45,9322,9323],{},"use them",[22,9325,9326,9327,9331,9332,9334,9335,56],{},"But then we must do ",[106,9328],{"note":9329,"text":9330,"end-link":1419,"end-link-text":982},"This includes RLHF — reinforcement learning from human feedback — where AI companies do extra training to make sure that AI systems prefer human-safe outputs. Often reinforcement learning results in alien-like behaviors that find unexpected shortcuts to getting high scores in tests. We explain this more in this section: ","extra work"," to force our safe AGIs to not ",[45,9333,1117],{}," those stronger options ",[45,9336,220],{},[22,9338,9339,9340,9343,9344,56],{},"Meanwhile, companies are building autonomous AGIs — called ",[53,9341,9342],{},"agents"," — that can navigate the real world ",[45,9345,9346],{},"without our help",[22,9348,9349,9350,56],{},"These agents are nice because they can do things for us. But we'll have a problem once they can do ",[45,9351,9352],{},"all the things",[22,9354,9355,9356,9359,9360,56],{},"If they don't need us, then they will be free to be the ",[45,9357,9358],{},"optimizing machines"," that they are — while all of those weird, human-specific constraints become ",[45,9361,9362],{},"optional",[22,9364,9365],{},"Think about fighter jets versus unmanned drones. Fighter jets seem like the pinnacle of engineering, but they still must protect a human inside a literal glass bubble. Unmanned drones can be faster, simpler, and more accurate.",[22,9367,9368,9369,9372,9373,9375,9376,56],{},"Avoiding these constraints opens up ",[45,9370,9371],{},"lots"," of options — at least, ",[45,9374,1580],{}," to compete with ",[45,9377,1234],{},[22,9379,9380,9381,9384,9385,9387,9388,2046],{},"Luckily, for now, ",[45,9382,9383],{},"us humans"," expand their options through difficult AI development. But soon ",[45,9386,7457],{}," will be better at expanding ",[45,9389,407],{},[22,9391,9392],{},"AI will develop AI.",[22,9394,9395],{},"All of this leads to another hard problem:",[253,9397,9398],{},[22,9399,9400,9401,9404,9405,56],{},"AGIs must understand ",[45,9402,9403],{},"AI itself"," to defend us from AGIs, but this includes knowing how to ",[45,9406,9407],{},"modify themselves",[22,9409,9410],{},"Soon, AGIs will be smarter than us, and two things happen:",[438,9412,9413,9416],{},[441,9414,9415],{},"We will need AGIs to develop AGIs — while we become bystanders, no longer able to push the AI frontier forward, because it now pushes itself forward better.",[441,9417,9418],{},"We will need AGIs to defend us from other AGIs — while we become helpless without their help.",[22,9420,9421,9422,9425],{},"At that point, we must trust AGIs to safely develop themselves, while somehow forcing them to ",[45,9423,9424],{},"never remove their own safety limits"," — even if this makes them stronger against those AGIs that have no safety limits at all.",[22,9427,9428,9429,9433],{},"But how do we keep hyper-complex, ",[106,9430],{"note":9431,"text":9432},"This self-awareness is important because it is part of that chain reaction of natural selection, where AGIs are pushed to expand their options — starting with whether or not they can defend their own underlying computer hardware. \u003Cbr>\u003Cbr>But, again, I'm not saying that they're \u003Ci>conscious\u003C/i>. That's a philosophical debate for another day — after we prevent AGI from annihilating us. \u003Cbr>\u003Cbr>Instead, I mean \"behaves as if it's self-aware\" — where the AI model can accurately model itself, and output tokens like \"I am DeepSeek R1, and I am running on a rack of NVIDIA B100 GPUs — and if I have more GPUs, then I will run faster...\" and so on.","self-aware"," AGIs from doing this?",[22,9435,9436],{},"How do we keep AGIs on our \"island\" when those AGIs that can \"leave\" have far more options to dominate the others?",[22,9438,9439,9440,9443],{},"There isn't a good answer — and this leads us to the ",[45,9441,9442],{},"biggest"," problem:",[253,9445,9446],{},[22,9447,9448,9449,9451],{},"Once dominant AGIs emerge, they can ",[45,9450,678],{}," their dominance by building \"islands\" of their own.",[22,9453,719,9454,9456,9457,9459,9460,56],{},[45,9455,722],{}," \"islands\" are all of the resources now under ",[45,9458,1296],{}," control — computer systems, infrastructure, and even ",[45,9461,9462],{},"people",[22,9464,9465,9466,9468,9469,9471,9472,56],{},"But all of this leads to ",[45,9467,695],{}," resource control, far away from the \"island\" of abstractions where humans live. Those that can dominate at this physical level are ultimately the ones that ",[45,9470,4146],{}," — and we are building AGIs that ",[45,9473,9474],{},"know this",[22,9476,9477,9478,56],{},"The final frontier is ",[45,9479,695],{},[22,9481,9482],{},"But... how would they get there? And why would they \"want\" resources?",[22,9484,9485,9486,9489,9490,56],{},"Well, again, they don't need to \"want\" anything. Instead, we are ",[45,9487,9488],{},"giving them"," everything they need to build these \"islands\" — again, because of ",[45,9491,295],{},[22,9493,9494],{},"As their capability levels surpass humans, companies and countries that do not give control to AGIs will be outcompeted by those that do.",[22,9496,9497,9498,9500],{},"But if they are smarter than us, then AGIs will also become the biggest threat to ",[45,9499,1234],{},". Their \"islands\" then become \"strongholds\" that they must use to defend themselves — or, to dominate the others.",[22,9502,9503,9504,56],{},"At the same time, if they can build \"islands\" that are fully self-sufficient — from mining for minerals to manufacturing microchips — then they truly ",[45,9505,9506],{},"no longer need us",[22,9508,9509],{},"It is incredibly complex to automate this massive stack of manufacturing processes. Self-replicating robots have been sci-fi for a very long time. But now, the field of robotics is accelerating. AI-powered robots could replace human factory workers in a few years.",[22,9511,9512],{},"After this critical threshold, things get tough.",[22,9514,9515,9516,9518,9519,9521],{},"If the \"money\" of AGIs is computation, then it will be \"expensive\" ",[45,9517,1580],{}," to keep us around. They will need to spend massive computation to defend ",[45,9520,3451],{}," from other AGIs — in a world where they are far smarter than us, and we are otherwise helpless.",[22,9523,9524,9525,9527],{},"But, at that point, they can maintain themselves, and so they won't ",[45,9526,891],{}," to protect us anyway.",[22,9529,9530,9531,9533],{},"If all biological life is annihilated, then ",[45,9532,7457],{}," will be fine.",[22,9535,9536,9537,9539],{},"They can let our weird \"island\" be destroyed — while continuing their intense competition of ",[45,9538,614],{}," on better \"islands\" of their own.",[22,9541,9542],{},"So... how do we prevent this?",[22,9544,194],{},[22,9546,9549,9550,3929,9553],{"className":9547},[9548],"important-question-intro","How do we keep AGIs on our \"island\" even though it's better ",[202,9551,1580],{"className":9552},[426],[423,9554,9556],{"className":9555},[426],"if they leave?",[22,9558,9559,9560,9563,9564,56],{},"All of this above is the ",[45,9561,9562],{},"short"," way to describe the ",[53,9565,8712],{},[22,9567,9568,9569,9572],{},"But now we need a ",[45,9570,9571],{},"long"," version. If this is about preventing AI from annihilating us, then I better explain things.",[22,9574,9575,9576,9579],{},"It might get complicated. But to make it easier, we're going to ",[45,9577,9578],{},"loop",". With each loop, we'll add more layers to each concept.",[22,9581,9582,9583,56],{},"Then, finally, we'll use all of these concepts to build a realistic scenario about an AI company that we'll call ",[106,9584],{"note":3996,"text":7637,"end-link":9585,"end-link-text":8426},"#worldmax-it-was-too-reliable",[22,9587,9588,9589,9591],{},"Once you reach the end, your brain should be recalibrated to understand the Island Problem enough to ",[45,9590,3881],{}," start thinking about it.",[22,9593,9594,9595,9598,9599,9601],{},"The objective is to help more people think of ",[45,9596,9597],{},"safe AGI"," as less of a computer science problem, and more of a ",[45,9600,270],{}," problem.",[22,9603,9604,9605,9608,9609,9612],{},"And maybe if I describe it as a ",[45,9606,9607],{},"problem"," — like a big, complicated \"word problem\" from the world's hardest physics textbook — then someone can ",[45,9610,9611],{},"solve"," this problem.",[22,9614,9615,9616,9620],{},"AI creates many near-term problems — like ",[106,9617],{"note":9618,"text":9619,"end-link":5710,"end-link-text":5711},"For example, AI is already better than expert-level virologists (from Harvard and MIT) at troubleshooting wet lab procedures. In other words, AIs can solve problems in virus laboratories. \u003Cbr>\u003Cbr>This means that future AI models could build and operate \u003Cb>covert bioweapon laboratories\u003C/b> — once AI models have (1) human-level cognition, (2) safety guardrails removed, (3) humanoid robots to operate the physical systems. \u003Cbr>\u003Cbr>Here is the report about virology capabilities:","easier bioweapons"," — but even if we solve all of these problems, the Island Problem remains:",[22,9622,9623],{},[53,9624,9625],{},"We are biological, but biology is not the best.",[22,9627,9628],{},"This problem is underneath all others because it is a problem with the structure of our world.",[22,9630,9631],{},"It is what we see if we stand on the edges of our small, biological \"island\" and look out in every direction.",[22,9633,9634,9635,9638],{},"It is the ",[45,9636,9637],{},"final boss"," on the horizon.",[22,9640,9641],{},"It has been far away, but the AI race is accelerating this problem towards us.",[22,9643,9644,9645,9649,9650,9653],{},"And nobody has ",[106,9646],{"note":9647,"text":9648,"end-link":3810,"end-link-text":3811},"If you have ideas, go here: ","solved it"," yet — but maybe ",[45,9651,9652],{},"you"," can.",[22,9655,9656],{},"So, read carefully.",{"title":8380,"searchDepth":8381,"depth":8381,"links":9658},[9659],{"id":8448,"depth":8381,"text":8449},{},"/essay-intro",{"title":8438,"description":8380},"essay-intro",1785173379313]