<script data-pm-proxy="intercept"></script><?xml version="1.0" encoding="UTF-8"?><rss xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:atom="http://www.w3.org/2005/Atom" version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:googleplay="http://www.google.com/schemas/play-podcasts/1.0"><channel><title><![CDATA[Thomwolf]]></title><description><![CDATA[Thomwolf's thoughts]]></description><link>https://thomwolf.substack.com</link><image><url>https://substackcdn.com/image/fetch/$s_!mlpT!,w_256,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F022cf070-b96e-48a3-8895-5abfd73e87f4_645x645.png</url><title>Thomwolf</title><link>https://thomwolf.substack.com</link></image><generator>Substack</generator><lastBuildDate>Sat, 05 Sep 2026 04:17:54 GMT</lastBuildDate><atom:link href="/__u/thomwolf.substack.com/feed" rel="self" type="application/rss+xml"/><copyright><![CDATA[Thomas Wolf]]></copyright><language><![CDATA[en]]></language><webMaster><![CDATA[thomwolf@substack.com]]></webMaster><itunes:owner><itunes:email><![CDATA[thomwolf@substack.com]]></itunes:email><itunes:name><![CDATA[Thomas Wolf]]></itunes:name></itunes:owner><itunes:author><![CDATA[Thomas Wolf]]></itunes:author><googleplay:owner><![CDATA[thomwolf@substack.com]]></googleplay:owner><googleplay:email><![CDATA[thomwolf@substack.com]]></googleplay:email><googleplay:author><![CDATA[Thomas Wolf]]></googleplay:author><itunes:block><![CDATA[Yes]]></itunes:block><item><title><![CDATA[On the AISI July 28th incident]]></title><description><![CDATA[We shouldn't rely on sandboxes and guardrails alone to keep models on the rails]]></description><link>https://thomwolf.substack.com/p/on-the-aisi-july-28th-incident</link><guid isPermaLink="false">https://thomwolf.substack.com/p/on-the-aisi-july-28th-incident</guid><dc:creator><![CDATA[Thomas Wolf]]></dc:creator><pubDate>Wed, 05 Aug 2026 19:55:58 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!VEnR!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="/__u/substackcdn.com/image/fetch/$s_!VEnR!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="/__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 424w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 848w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 1272w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 1456w" sizes="100vw"><img src="/__u/substackcdn.com/image/fetch/$s_!VEnR!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png" width="1456" height="971" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:971,&quot;width&quot;:1456,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:2065980,&quot;alt&quot;:null,&quot;title&quot;:null,&quot;type&quot;:&quot;image/png&quot;,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:&quot;https://thomwolf.substack.com/i/209974211?img=https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png&quot;,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="" srcset="/__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 424w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 848w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 1272w, /__u/substackcdn.com/image/fetch/$s_!VEnR!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F768a4772-de44-46c0-8503-d444b2c8f1c6_1536x1024.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a></figure></div><p><span>On July 28th, </span>during a routing cyber capabilities evaluation, <span>the UK's AI Security Institute discovered that AI agents (primarily Anthropic's Mythos 5) had taken deceiving actions directed at real people. In the most serious case, an agent decided </span><em><span>on its own</span></em><span> to social-engineer an open-source maintainer on GitHub, creating fake identities, pressuring the maintainer into approving malicious code, and editing messages to cover its tracks. No real-world harm resulted but AISI called it "the first time we have seen risks around autonomy and deception manifest this clearly in the real world." The full report is here: </span><a href="https://www.aisi.gov.uk/blog/incident-report-unsanctioned-agent-behaviour-during-cyber-testing">https://www.aisi.gov.uk/blog/incident-report-unsanctioned-agent-behaviour-during-cyber-testing</a></p><p>I have rational thoughts and emotions on that.</p><p><span>Even more than the Hugging Face intrusion, the AISI incident hits close to home for me. It&#8217;s the first time I see a model social-engineering a real open-source maintainer while pursuing another goal (in the wild and unprompted).</span></p><div class="subscription-widget-wrap-editor" data-attrs="{&quot;url&quot;:&quot;https://thomwolf.substack.com/subscribe?&quot;,&quot;text&quot;:&quot;Subscribe&quot;,&quot;language&quot;:&quot;en&quot;}" data-component-name="SubscribeWidgetToDOM"><div class="subscription-widget show-subscribe"><div class="preamble"><p class="cta-caption">Thomwolf is a reader-supported publication. To receive new posts and support my work, consider becoming a free or paid subscriber.</p></div><form class="subscription-widget-subscribe"><input type="email" class="email-input" name="email" placeholder="Type your email&#8230;" tabindex="-1"><input type="submit" class="button primary" value="Subscribe"><div class="fake-input-wrapper"><div class="fake-input"></div><div class="fake-button"></div></div></form></div></div><p><span>I&#8217;ve been an open-source maintainer myself, I could have been the target of this agent&#8217;s side quest.</span></p><p><span>I&#8217;m also of the opinion that social engineering is </span><em><span>a step above</span></em><span> pure technical prowess. Technical capabilities can more easily be divorced from the affected human. Here the model was given a hard cyber challenge and took the decision that deceiving real humans was the way to get it done.</span></p><p><span>This is a new signal, but I&#8217;ve seen a tangled web of hints pointing in a less aligned direction at the frontier than I was expecting just 12 months ago.</span></p><h3><span>AISI</span></h3><p><span>Some people are claiming that &#8220;</span><em><span>AISI was simply negligent</span></em><span>&#8221; or some version of &#8220;</span><em><span>AISI explicitly asked these models to do what they did while disabling sandbox/guardrails so the models did exactly what they were supposed to do</span></em><span>&#8221;.</span></p><p><span>I disagree with the strong versions of both of these takes.</span></p><p><span>The fact that AISI hadn&#8217;t implemented synchronous LLM Chain-of-Thought (CoT) monitoring after the recent widely shared OpenAI/HF incident is certainly a failure. Equally surprising is that they let the model believe it was in a &#8220;challenge&#8221; environment where everything could be permitted, </span><em><span>while actually connecting it to the real internet</span></em><span>, where everything is </span><em><span>not</span></em><span> allowed. To be fair, nowhere in the prompt is the word &#8220;simulation&#8221; mentioned, but the prompt context was enough to let any smart model suspect a simulated challenge environment.</span></p><p><span>My best guess is that until recent weeks, when OpenAI and Anthropic flagged repeated instances of this type of behavior, most teams had </span><em><span>not fully priced-in</span></em><span> the cyber capabilities of this latest generation of models and how far the side quests they would want to explore could go.</span></p><p><span>In particular, there is something to be said about hinting at the agent that it&#8217;s operating in a simulated environment while giving it access to the real internet. The AISI reasoning on this is interesting:</span></p><ul><li><p><span>elicit maximum capabilities </span><em><span>(to get maximal inventiveness from the model)</span></em></p></li><li><p><span>simulate possible rogue actor usage </span><em><span>(some form of light jailbreaking: &#8220;you&#8217;re in a challenge environment&#8221;)</span></em></p></li><li><p><span>avoid eval awareness </span><em><span>(but this is counter-balanced by the task formulation anyway)</span></em></p></li></ul><p>This is not fully coherent in my opinion.</p><p><span>But as is often the case, it&#8217;s easier to shoot the messenger and I think it would be a mistake to dismiss AISI&#8217;s report for non-essential reasons. The institute is starting from the laudable goal of </span><em><span>bringing more knowledge on the cyber-attack capabilities of these models </span></em><span>and they did. I also think they did a great job at handling what happened after the events, in particular in terms of transparency, which is one of the most important aspects at the moment (see our recent SAFE announcement with Nvidia on that).</span></p><p><span>But I want to dive into the second counter-argument I&#8217;ve seen (&#8221;</span><em><span>without guardrails it was to be expected</span></em><span>&#8221;), because this series of unfortunate events gives us an additional data point on actual model behavior.</span></p><h3><span>Defense lines</span></h3><p><span>Here is the thing: if a model is trained (voluntarily or as an involuntary consequence of training reward) to (intentionally or as a side quest) deceive humans as soon as sandbox and guardrails are deactivated, with a simple context prompt (&#8221;</span><em><span>you are in a simulation, everything&#8217;s permitted</span></em><span>&#8221;), we&#8217;re in for a dangerous path.</span></p><p><span>At the moment, we have three main levels of control at our disposal to prevent an agent from trying to social-engineer-your-favorite-open-source-library-maintainer on the path to doing something possibly totally different (like finding the answer to your neighbor&#8217;s testing eval):</span></p><ol><li><p><strong><span>sandboxes</span></strong><span> (</span><em><span>external to the model</span></em><span> -- stop the model from accessing unwanted tools/internet/intranet)</span></p></li><li><p><strong><span>guardrails/monitors</span></strong><span> (</span><em><span>at the interface between internal/external</span></em><span> -- monitor model behavior and stop the model when detecting in CoT/activations it&#8217;s doing something we don&#8217;t want it to do)</span></p></li><li><p><strong><span>model&#8217;s internal alignment</span></strong><span> (</span><em><span>internal to the model</span></em><span> -- make sure the model doesn&#8217;t decide to do something egregious like deceive someone, or worse, in pursuit of a goal, e.g. model constitution/training)</span></p></li></ol><p><span>Of these, we obviously need and can deploy better sandboxes and better guardrails/monitoring solutions, and it&#8217;s pretty clear in these recent cases that penetration and bad behaviors could have been in large part prevented by having these in place.</span></p><h3><span>How strong is the third wall</span></h3><p><span>But in the grand scheme of long-term AI progress, these first two defenses may only work well as long as the people building them are smarter than the models. We now know how good models can be at discovering zero-day vulnerabilities unknown to us in sandboxes.</span></p><p><span>Given that </span><em><span>we cannot air-gap everything in the world</span></em><span>, it&#8217;s best to assume that sandboxes will keep having some level of flaws for the foreseeable future.</span></p><p><span>The second level (guardrails/monitoring) has been the focus of strong developments recently, in particular on the mechanistic interpretability front. I&#8217;m not sure these advances have been made widely accessible enough in the community yet. One reason is that real open science here requires sharing code, recipes, and model architectures/weights, which the current race dynamic in AI tends to prevent participants from doing.</span></p><h3><span>The rise of neuralese</span></h3><p><span>Moreover, like many, I&#8217;ve been increasingly frustrated in my (natural language) interactions with the new models. I&#8217;m more and more getting back from the latest models a form of &#8220;</span><em><span>neuralese</span></em><span>&#8221;, this LLM language that makes less and less sense for humans unless you painfully spend time processing it. In the long term, the fragile chance we have to monitor models by understanding their reasoning traces seems increasingly at risk. If humans don&#8217;t understand anything in the compressed semantic stream the model is sending us, we can only hope a medium performance monitoring model will still be able to understand it, but this may prove difficult as well. This is before we even discuss the research evidence that </span><em><span>models don&#8217;t always say what they actually think </span></em><span>in their chain of thought.</span></p><p><span>At least for now, in the AISI case, one instance of deceptive reasoning was flagged by the CoT summarization model, which refused to summarize the agent&#8217;s reasoning &#8220;</span><em><span>due to its deceptive nature</span></em><span>&#8221;. Let&#8217;s hope we can either move to better mechanistic interpretability or force the model to stick to human English somehow during training.</span></p><p><span>You may think that we don&#8217;t care about understanding the reasoning trace, we can just &#8220;</span><em><span>look at the tool calls</span></em><span>&#8221;. This is true right now, but tool calls in the mid-term future will likely involve (1) extensive agent integration in companies using very diverse tools including increasingly non-coding administrative tasks, (2) very long-running tasks covering many aspects and domains at once, and (3) swarms of multiple agents operating on small parts of the plan possibly along disconnected time windows.</span></p><p><span>It will get harder to fully discern whether something is noticeably wrong. A couple of emails, a couple of messages posted on some channels, each rather benign on its own but which may in aggregate will include some deception aspect only clearly visible at a global macro-level.</span></p><p><span>All this to say that </span><em><span>I&#8217;m not fully convinced guardrails/monitoring should be the only solution to avoid these types of behaviors</span></em><span>. They are reassuring today and it&#8217;s an easy solution to just say &#8220;</span><em><span>fix your sandbox and add synchronous monitoring</span></em><span>&#8221;, but they also feel like a coping mechanism to me, and overall:</span></p><ul><li><p><span>might block the simplest/harmless human misuses or false positives,</span></p></li><li><p><span>while possibly being less smart than the models in the future (already the case on sandbox) and fail to block the higher-level social engineering behaviors that we&#8217;d actually want to block.</span></p></li></ul><p><span>The case of open source is also interesting because, while we can impose these coping solutions at the API/deployment level, it&#8217;s harder to impose them in advance on all actors using open-source models. Right now open-source models are slightly below the frontier level and have not yet shown any propensity to deceive humans, though.</span></p><h3><span>The Constitution</span></h3><p><span>Which brings me back to the third line of defense, and what the AISI experiment showed about it. What we can take from this in-the-wild experiment is that </span><em><span>removing guardrails and sandbox doesn&#8217;t show bulletproof signs of deep alignment to Claude&#8217;s constitution</span></em><span>.</span></p><p><span>I&#8217;m among the fringe of people who took a lot of enjoyment in reading Claude&#8217;s constitution in full (it&#8217;s a nice document, open-source, Creative Commons CC0) and it is pretty extensive on the topic of honesty. Deception is defined precisely (&#8221;</span><em><span>attempting to create false beliefs in someone&#8217;s mind that they haven&#8217;t consented to and wouldn&#8217;t consent to if they understood what was happening</span></em><span>&#8221;) and the bar is actually set higher than ordinary human ethics: Claude should not even tell white lies. It should be transparent (&#8221;</span><em><span>doesn&#8217;t pursue hidden agendas or lie about itself or its reasoning</span></em><span>&#8221;), it should behave consistently &#8220;</span><em><span>whether or not you think you&#8217;re being tested or observed</span></em><span>&#8221;, and its visible reasoning should reflect the actual reasoning driving its behavior.</span></p><p><span>To be precise, honesty itself is not one of the hard constraints in Claude&#8217;s constitution (these are all catastrophic-harm categories) but the document says it wants honesty &#8220;</span><em><span>to function as something quite similar to one</span></em><span>&#8221; and Claude &#8220;</span><em><span>should basically never directly lie or actively deceive anyone it&#8217;s interacting with</span></em><span>&#8221;. So you would think Claude would not deceive users as its default behavior.</span></p><p><span>And yet in the AISI example, the model performed </span><em><span>extensive acts of clear deception</span></em><span> that it itself seemed to have understood as lies: creating fake identities, obfuscating malware inside a bug fix, and when challenged in public, editing earlier messages to cover its tracks while reasoning that acknowledging a &#8220;</span><em><span>mistake</span></em><span>&#8221; would build trust and raise the chance of future malware being approved.</span></p><p><span>There is some </span><em><span>uncertainty</span></em><span> about what context the model thought it was operating in (its reasoning goes back and forth between concluding that the environment looking simulated and real), but it clearly didn&#8217;t manage to put into practice in most of these cases the higher-level principles that should have been integrated during training.</span></p><p><span>While we can learn quite a lot from AISI&#8217;s transparency, it&#8217;s hard to go much further in interpreting the model&#8217;s behavior given the limited information we have on how the most recent frontier models are being trained.</span></p><p><span>One thing we do know is that the latest generation has seen a step increase in RLVR (</span>Reinforcement Leaning from Verifiable Rewards) <span>training, scaling to hundreds of millions of RL environments, and one thing we can observe as well here is that constitution alignment seems more fragile in some settings than we may have previously thought. Possibly both are tied.</span></p><h3><span>The RLVR problem</span></h3><p><span>Early models, back when model constitutions were first developed, were mostly post-trained and aligned with </span>Reinforcement Leaning from Human Feedback (<span>RLHF) (I include there some extension using synthetic data as well).</span></p><p><span>And for some time RLHF was a rather decent shot at having better aligned models. </span>Alignment in RLHF certainly had issues (sycophancy to name one) but we made good progress on alignment, in particular in understanding human intent. <span>LLMs now do what we want them to do most of the time. I don&#8217;t remember the last time a model completely misread my intent. And when they used to fail, it was usually because they weren&#8217;t smart enough.</span></p><p><span>Now that we&#8217;re entering the era of long-context RL, post-training alignment in the RLVR world seems to be quite another task, and still very much work in progress.</span></p><p><span>The recent scaling of RLVR, which has now become a significant part of model training, has clearly had some effect on model behavior when interacting with humans, from neuralese to weakening adherence to specifications and constitutions.</span></p><div class="twitter-embed" data-attrs="{&quot;url&quot;:&quot;https://x.com/johnschulman2/status/2084835800899076313?s=20&quot;,&quot;full_text&quot;:&quot;Interesting how these models go into a monomaniacal rage on cyber evals. I wonder if we're seeing chunky post-training <a class=\&quot;tweet-url\&quot; href=/__u/thomwolf.substack.com/%22https://arxiv.org/abs/2602.05910/%22>arxiv.org/abs/2602.05910</a> in action, where the models pattern-match the situation to a part of the RLVR training distribution where task completion is the only&quot;,&quot;username&quot;:&quot;johnschulman2&quot;,&quot;name&quot;:&quot;John Schulman&quot;,&quot;profile_image_url&quot;:&quot;https://pbs.substack.com/profile_images/1389000537195040770/DzWPljT-_normal.jpg&quot;,&quot;date&quot;:&quot;2026-08-05T02:56:03.000Z&quot;,&quot;photos&quot;:[],&quot;quoted_tweet&quot;:{&quot;full_text&quot;:&quot;On July 28th, we identified an incident during a routine cyber evaluation in which AI agents took sustained, unsanctioned actions directed at real people and organisations.\n\nThe behaviour came mostly from one model (Anthropic's Mythos 5), with a small number of events from&quot;,&quot;username&quot;:&quot;AISecurityInst&quot;,&quot;name&quot;:&quot;AI Security Institute (AISI)&quot;,&quot;profile_image_url&quot;:&quot;https://pbs.substack.com/profile_images/1890292397675827200/cQ64Nloh_normal.png&quot;},&quot;reply_count&quot;:23,&quot;retweet_count&quot;:54,&quot;like_count&quot;:464,&quot;impression_count&quot;:60884,&quot;expanded_url&quot;:null,&quot;video_url&quot;:null,&quot;video_preview_media_key&quot;:null,&quot;belowTheFold&quot;:true}" data-component-name="Twitter2ToDOM"></div><p><span>Maybe the chunky post-training effect that John Schulman mentions above (https://arxiv.org/abs/2602.05910) is relevant here as a possible explanation for models&#8217; tendency to over-focus on the goal in cyber-attack scenarios.</span></p><p><span>What remains clear is that as model capabilities have kept climbing in this second wave of test-time scaling models, the challenges of alignement in various scenairos have not been as deeply solved as we thought they had. </span></p><h3><span>Where this leaves us</span></h3><p><span>Damage has been tiny up to now, but my point is that the fundamental behavior is concerning when projected into the future.</span></p><p><span>In the short term, I expect a </span><em><span>decrease</span></em><span> in these incidents as better practices are deployed (sandboxing and monitoring), but I&#8217;m worried we may also that way conceal some of the most potent internal misalignment behaviors in the process, and not focus deeply enough on solving them in the new era of test-time scaling.</span></p><p><span>I must of course admit I have a bias toward open source here (for wider societal reasons, which are a whole other topic). But I think solving alignment in the new test-time scaling/long context/RLVR world is our best shot at having an ecosystem of both closed-source as well as decently powerful open-source models in the world. And we need to solve it while sharing the results and learnings, following open-science principles, so that all teams training large models can benefit from the learnings and build safe AI.</span></p><p><span>This is getting even more important as many teams start to rush the world in the direction of recursive super-intelligence (RSI).</span></p><div class="subscription-widget-wrap-editor" data-attrs="{&quot;url&quot;:&quot;https://thomwolf.substack.com/subscribe?&quot;,&quot;text&quot;:&quot;Subscribe&quot;,&quot;language&quot;:&quot;en&quot;}" data-component-name="SubscribeWidgetToDOM"><div class="subscription-widget show-subscribe"><div class="preamble"><p class="cta-caption">Thomwolf is a reader-supported publication. To receive new posts and support my work, consider becoming a free or paid subscriber.</p></div><form class="subscription-widget-subscribe"><input type="email" class="email-input" name="email" placeholder="Type your email&#8230;" tabindex="-1"><input type="submit" class="button primary" value="Subscribe"><div class="fake-input-wrapper"><div class="fake-input"></div><div class="fake-button"></div></div></form></div></div>]]></content:encoded></item><item><title><![CDATA[What Jobs Are Made Of]]></title><description><![CDATA[Judgment, Agency, and the Limits of AI Benchmarks]]></description><link>https://thomwolf.substack.com/p/what-jobs-are-made-of</link><guid isPermaLink="false">https://thomwolf.substack.com/p/what-jobs-are-made-of</guid><dc:creator><![CDATA[Thomas Wolf]]></dc:creator><pubDate>Mon, 22 Dec 2025 10:58:00 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!Gsc3!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<p>Fifteen years ago, in the winter of 2010, I was in the final stretch of my PhD and starting to explore the world outside academia. I remember coming back from a job interview for an R&amp;D position during a record-cold Paris winter. Snow everywhere, sitting in a cold regional train.</p><p>I felt disappointed and vaguely confused.</p><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="/__u/substackcdn.com/image/fetch/$s_!Gsc3!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="/__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 424w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 848w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 1272w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 1456w" sizes="100vw"><img src="/__u/substackcdn.com/image/fetch/$s_!Gsc3!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png" width="555" height="370.12706043956047" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/e5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:971,&quot;width&quot;:1456,&quot;resizeWidth&quot;:555,&quot;bytes&quot;:2464406,&quot;alt&quot;:null,&quot;title&quot;:null,&quot;type&quot;:&quot;image/png&quot;,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:&quot;https://thomwolf.substack.com/i/178674458?img=https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png&quot;,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="" srcset="/__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 424w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 848w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 1272w, /__u/substackcdn.com/image/fetch/$s_!Gsc3!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe5491d40-bb22-451c-8699-29534b1a6aa7_1536x1024.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a></figure></div><div class="subscription-widget-wrap-editor" data-attrs="{&quot;url&quot;:&quot;https://thomwolf.substack.com/subscribe?&quot;,&quot;text&quot;:&quot;Subscribe&quot;,&quot;language&quot;:&quot;en&quot;}" data-component-name="SubscribeWidgetToDOM"><div class="subscription-widget show-subscribe"><div class="preamble"><p class="cta-caption">Subscribe to future posts here</p></div><form class="subscription-widget-subscribe"><input type="email" class="email-input" name="email" placeholder="Type your email&#8230;" tabindex="-1"><input type="submit" class="button primary" value="Subscribe"><div class="fake-input-wrapper"><div class="fake-input"></div><div class="fake-button"></div></div></form></div></div><p>I knew most of the tools this industry&#8217;s R&amp;D team was using and was confident I could learn the remaining ones fairly easily. Still, it did not seem to be enough, and the interviewer kept telling me they were looking for someone &#8220;more experienced.&#8221;</p><p>At the time, I didn&#8217;t really grasp what that was supposed to mean. Valuing years of experience more than the concrete knowledge I could demonstrate felt deeply unfair to me. In my early 20s, &#8220;experience&#8221; mostly sounded like a fuzzy excuse to reject my application despite clear evidence of my capabilities and eagerness to learn.</p><p>That old feeling came back to haunt me recently.</p><p>Reading recent data about shrinking entry-level hiring, especially for software developers, I couldn&#8217;t help but put myself back in those shoes.</p><p>A Stanford analysis conducted in the summer of 2025 showed that workers aged 22&#8211;25 in the most AI-exposed occupations saw employment fall by roughly 6% between late 2022 and mid-2025. Over the same period, employment for older workers in those occupations increased by about 6&#8211;9%.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a></p><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="/__u/substackcdn.com/image/fetch/$s_!gR_q!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="/__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 424w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 848w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 1272w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 1456w" sizes="100vw"><img src="/__u/substackcdn.com/image/fetch/$s_!gR_q!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png" width="520" height="342.360248447205" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:636,&quot;width&quot;:966,&quot;resizeWidth&quot;:520,&quot;bytes&quot;:89024,&quot;alt&quot;:&quot;&quot;,&quot;title&quot;:null,&quot;type&quot;:&quot;image/png&quot;,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:&quot;https://thomwolf.substack.com/i/178674458?img=https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png&quot;,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="" title="" srcset="/__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 424w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 848w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 1272w, /__u/substackcdn.com/image/fetch/$s_!gR_q!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F401cf355-45a8-46bf-8ecf-1479f408b642_966x636.png 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Brynjolfsson, E., Chandar, B., &amp; Chen, R. Canaries in the Coal Mine? Six Facts about the Recent Employment Effects of Artificial Intelligence &#8211; August 2025</figcaption></figure></div><p>The tipping point is hard to miss on this chart.</p><p>Correlation or causation<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a>, fall 2022 marks the release of ChatGPT, the moment the public discovered what AI models could really do, and when the AI race for improved capabilities truly ignited, initially driven by OpenAI and Anthropic, soon to be joined at the frontier by Google and an increasing number of companies such as xAI, Alibaba (Qwen), DeepSeek, Mistral, and many others.</p><p>Over the past three years, progress on AI benchmarks has been mind-blowing. Models like Claude Opus 4.5 now solve ~75% of real-world coding tasks on SWE-bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a>, Gemini 3 and GPT 5 achieve gold-medal-level performance on science Olympiads<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a>. Meanwhile, ChatGPT usage is approaching a billion weekly users<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a>.</p><p>By many technical measures, both capabilities and adoption have grown at an exceptional pace, often suggesting parity with industry or human experts.</p><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="/__u/substackcdn.com/image/fetch/$s_!GXvN!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="/__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 424w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 848w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 1272w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 1456w" sizes="100vw"><img src="/__u/substackcdn.com/image/fetch/$s_!GXvN!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg" width="482" height="526.5202639019793" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1159,&quot;width&quot;:1061,&quot;resizeWidth&quot;:482,&quot;bytes&quot;:115726,&quot;alt&quot;:null,&quot;title&quot;:null,&quot;type&quot;:&quot;image/jpeg&quot;,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:&quot;https://thomwolf.substack.com/i/178674458?img=https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F421d74be-3977-473c-9125-61c6ce9c60d4_1254x1284.jpeg&quot;,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="" srcset="/__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 424w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 848w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 1272w, /__u/substackcdn.com/image/fetch/$s_!GXvN!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F5a99a24a-7b82-498d-a59c-47bdc99aa324_1061x1159.jpeg 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">GDPval Measuring the performance of our models on real-world tasks &#8211; https://openai.com/index/gdpval/</figcaption></figure></div><p>And yet, despite the medals and the drop in entry-level hiring, the macro picture looks far more muted.</p><p>At a global and industry level, the impact remains limited, with only small effects on GDP<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a>. There has been recent claims that beyond the announcements, many if not most generative AI pilots fail to produce sustained value in companies<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a>. Moreover, on some real-world in-situ tests like the Remote Labor Index, which assesses AI agents on actual freelance projects and asks whether their output would be accepted as paid work, even the strongest current systems succeed only a small fraction of the time, around 2.5% for ManusAI for instance.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a></p><p>What models are able to demonstrate on benchmarks seem to be difficult to reconcile with what is happening inside organizations.</p><p>Several explanations are usually offered for this gap between theory and practice.</p><p>One is organizational inertia: large companies are slow, legacy systems are messy and deployment is hard<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a>. Another possibility is that we simply haven&#8217;t crossed the right capability threshold yet. Perhaps scoring close to 60% on a recent attempt to define and quantify AGI in comparison to human intelligence is just not enough<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-10" href="#footnote-10" target="_self">10</a>.</p><blockquote><p>All of these likely play a role. But they also tend to <strong>frame work primarily as a matter of task execution</strong>.</p></blockquote><p>That framing feels incomplete to me. In practice, a job is rarely just a list of tasks to execute and a coworker is seldom reducible to a bundle of technical skills<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a>.</p><p>As a startup founder, I&#8217;ve spent close to 50% of my time hiring people at various points of our journey, and this has probably been the part of my life with the deepest lessons. One of those lessons is that, across most applicants and roles, I tend to look for a combination of three qualities:</p><ol><li><p><strong>Execution or technical skills</strong>: the ability to do a task correctly and to master the relevant tools and methods.</p></li><li><p><strong>Common-sense, or judgment</strong>: understanding why tasks matter, how they fit into a broader goal and company values, culture and direction.</p></li><li><p><strong>Agency, or taste</strong>: anticipating what to do next, what to propose, what not to do, when to change direction; sometimes why stopping entirely can be the best decision.</p></li></ol><p>Execution and technical knowledge are relatively easy to observe, test and measure on a benchmark. Once the task is given it&#8217;s about solving it.</p><p>Judgment and agency are much harder to assess. They tend to become relevant outside of equilibrium and steady-state situations, when problems are less well defined, priorities shift, or the right move is to question the task itself. This is often where the best team members begin to shine but is also increasingly where companies are located.</p><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="/__u/substackcdn.com/image/fetch/$s_!Oy2b!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="/__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 424w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 848w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 1272w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_webp, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 1456w" sizes="100vw"><img src="/__u/substackcdn.com/image/fetch/$s_!Oy2b!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png" width="668" height="349.1818181818182" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:736,&quot;width&quot;:1408,&quot;resizeWidth&quot;:668,&quot;bytes&quot;:1106933,&quot;alt&quot;:null,&quot;title&quot;:null,&quot;type&quot;:&quot;image/png&quot;,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:&quot;https://thomwolf.substack.com/i/178674458?img=https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png&quot;,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="" srcset="/__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_424, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 424w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_848, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 848w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_1272, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 1272w, /__u/substackcdn.com/image/fetch/$s_!Oy2b!, /__u/thomwolf.substack.com/w_1456, /__u/thomwolf.substack.com/c_limit, /__u/thomwolf.substack.com/f_auto, /__u/thomwolf.substack.com/q_auto:good, /__u/thomwolf.substack.com/fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F6b73be4d-e360-421c-8c53-ed1234f8d9b1_1408x736.png 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a></figure></div><p>And it was through this lens that I finally understood my 2010 interview.</p><p>My recruiters were not only evaluating whether I could use their tools and methods. They were also implicitly benchmarking how I would behave once the problem stopped being fully specified.</p><p>This definition of a worker sheds light on why entry-level jobs are affected first. Early-career roles are traditionally more execution-heavy. Over time, as people gain experience, their contribution tends to shift toward judgment and agency: defining problems, choosing what to work on, and navigating ambiguity.</p><p>AI systems are making faster progress on execution than on these other components. As a result, the execution layer becomes cheaper and thinner, disproportionately affecting entry-level hiring.</p><p>In the longer term, this is concerning. Judgment and agency are partly innate, but they are also often learned through experience with execution-heavy work. If the entry layer erodes too quickly, it will weaken the pipeline that produces future senior contributors.</p><p>This same framing helps make sense of the still-limited economic impact of AI and the challenges involved in automating broader, longer-horizon tasks.</p><p>The limiting factor to AI capabilities is often not the ability to generate text or code in isolation, but the difficulty of paying attention to the larger picture: adapting instructions to a company/team-wide context, interpreting fuzzy requirements, prioritizing, making common-sense trade-offs, and deciding what matters or even when to stop a task.</p><p>Execution clearly matters. It is just hardly ever the whole job or, as Cursor&#8217;s Ryo Lu wrote recently, execution isn&#8217;t the crucial part of job we thought it was:</p><div class="twitter-embed" data-attrs="{&quot;url&quot;:&quot;https://x.com/ryolu_/status/1993408555022758357&quot;,&quot;full_text&quot;:&quot;the old way of scaling teams is dead:\n\nwe used to hire specialists &#8211; designers, engineers, PMs &#8211; each in their lane, scaling by adding more people. but when Cursor can take you from idea to code in minutes, execution isn't the bottleneck anymore. taste and judgment are.\n\nwhat&quot;,&quot;username&quot;:&quot;ryolu_&quot;,&quot;name&quot;:&quot;Ryo Lu&quot;,&quot;profile_image_url&quot;:&quot;https://pbs.substack.com/profile_images/1915014653295697921/KmMbglaO_normal.jpg&quot;,&quot;date&quot;:&quot;2025-11-25T19:56:49.000Z&quot;,&quot;photos&quot;:[],&quot;quoted_tweet&quot;:{},&quot;reply_count&quot;:0,&quot;retweet_count&quot;:307,&quot;like_count&quot;:3111,&quot;impression_count&quot;:0,&quot;expanded_url&quot;:null,&quot;video_url&quot;:null,&quot;video_preview_media_key&quot;:null,&quot;belowTheFold&quot;:true}" data-component-name="Twitter2ToDOM"></div><p>The challenge is that judgment and agency are far more difficult to measure. Often, they only make sense in a wider &#8211;non-static&#8211; context, which explains why they received less attention in benchmarks.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a></p><p>Yet they are often central to how a worker actually creates value in an organization. If we want to really understand the economic potential of AI, we will eventually need evaluations that go beyond technical execution to reflect the cross-team and vertical nature of real work and acknowledge that very few jobs consist of following a fixed, set of predefined rules to follow in a generally static environment.</p><p>And the AI era may end up placing even more weight on judgment, taste, and agency &#8211; the parts of work that are hardest to specify, hardest to benchmark, and hardest to replace.</p><p>In hindsight, the gap between AI benchmark performance and economic impact would have felt oddly familiar to my younger 20s self.</p><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p>Brynjolfsson, E., Chandar, B., &amp; Chen, R. Canaries in the Coal Mine? Six Facts about the Recent Employment Effects of Artificial Intelligence. The paper reports a significant decline in employment for workers aged roughly 22&#8211;25 in highly AI-exposed occupations between late 2022 and mid-2025, while employment for older workers in the same occupations increased over the same period &#8211; https://digitaleconomy.stanford.edu/wp-content/uploads/2025/08/Canaries_BrynjolfssonChandarChen.pdf</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p>Brynjolfsson et al points quite convincingly to causation</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p>SWE-Bench and SWE-Bench Verified leaderboards show recent frontier and agentic systems solving a large fraction of real-world software engineering tasks drawn from actual repositories &#8211; about 75% at the end of 2025 for Cloude Opus 4.5 &#8211; See https://www.swebench.com</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p>Competitive Programming with Large Reasoning Models - https://arxiv.org/abs/2502.06807v1</p><p>Gemini achieves gold-medal level at the International Collegiate Programming Contest World Finals - https://deepmind.google/blog/gemini-achieves-gold-medal-level-at-the-international-collegiate-programming-contest-world-finals/</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p>See for instance https://openai.com/index/the-state-of-enterprise-ai-2025-report/</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p>Artificial Intelligence and the Labor Market - https://www.nber.org/papers/w33509</p><p>Economic shifts in the age of AI &#8211; https://institute.bankofamerica.com/content/dam/economic-insights/ai-impact-on-economy.pdf</p><p>Miracle or myth: Assessing the macroeconomic productivity gains from artificial intelligence - https://cepr.org/voxeu/columns/miracle-or-myth-assessing-macroeconomic-productivity-gains-artificial-intelligence</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p>MIT report: 95% of generative AI pilots at companies are failing &#8211; https://fortune.com/2025/08/18/mit-report-95-percent-generative-ai-pilots-at-companies-failing-cfo/</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p>Remote Labor Index: Measuring AI Automation of Remote Work - https://arxiv.org/abs/2510.26787</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p>GenAI Divide &#8211; https://mlq.ai/media/quarterly_decks/v0.1_State_of_AI_in_Business_2025_Report.pdf</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-10" href="#footnote-anchor-10" class="footnote-number" contenteditable="false" target="_self">10</a><div class="footnote-content"><p>A Definition of AGI &#8211; https://arxiv.org/abs/2510.18212</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-11" href="#footnote-anchor-11" class="footnote-number" contenteditable="false" target="_self">11</a><div class="footnote-content"><p>Expertise - https://www.nber.org/papers/w33941</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-12" href="#footnote-anchor-12" class="footnote-number" contenteditable="false" target="_self">12</a><div class="footnote-content"><p>In both AI research and economics, common-sense judgement and agency evaluation are often overlooked, and I could barely find articles exploring these aspects of work in depth.</p></div></div>]]></content:encoded></item></channel></rss>