[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"sanity-sRFEDBsot21yNASgUkD57xFOTirTJZHlQJ_-zp1ER70":3,"sanity-soaLIrlcnMn5ggvLvieKJLH79sNHgq5Zi4VJtITiETs":674},{"data":4,"sourceMap":-1},{"latestPodcast":5,"latestReleases":14,"post":39,"recent":649},[6],{"_id":7,"publishedAt":8,"slug":9,"sponsored":12,"title":13},"8e6c3a8a-3d27-44c8-be90-0b74d1090a4a","2026-10-06T17:00:00.000Z",{"_type":10,"current":11},"slug","tales-from-the-2026-developer-survey-results",null,"Tales from the 2026 Developer Survey results",[15,21,27,33],{"_id":16,"publishedAt":17,"slug":18,"title":20},"7a2d88e5-ee53-4f7c-a46e-47bed1cebadf","2026-09-30T16:00:00.000Z",{"_type":10,"current":19},"anyone-can-start-building-verified-knowledge-with-stack-internal","Anyone can start building verified knowledge with Stack Internal",{"_id":22,"publishedAt":23,"slug":24,"title":26},"12c6a8a7-135f-401f-ac1a-1c26c33e69c0","2026-09-03T16:00:00.000Z",{"_type":10,"current":25},"security-control-and-accessibility-si-2026-6","Elevating security, control, and accessibility: Stack Internal 2026.6",{"_id":28,"publishedAt":29,"slug":30,"title":32},"adcf1bca-3295-4ac5-9b3c-23f337974190","2026-07-30T15:10:00.000Z",{"_type":10,"current":31},"introducing-stack-internal-new-platform-experience","Your trusted knowledge layer: Introducing Stack Internal's new platform experience",{"_id":34,"publishedAt":35,"slug":36,"title":38},"eb5b66eb-9410-4329-83bb-22bbff39402a","2026-04-28T13:00:00.000Z",{"_type":10,"current":37},"turn-scattered-knowledge-into-trusted-intelligence","Turning scattered knowledge into trusted intelligence: Stack Internal 2026.3",{"_createdAt":40,"_id":41,"_rev":42,"_system":43,"_type":47,"_updatedAt":48,"author":49,"body":67,"comments":617,"dateUrl":618,"excerpt":619,"featureStackTech":617,"image":620,"product":12,"publishedAt":624,"slug":625,"sponsored":12,"tags":627,"title":648,"visible":617},"2026-10-06T14:41:53Z","358fa8ac-2467-4ef4-893c-dad3490046df","WHbP4dBx2d4iMWuzd7fP4p",{"base":44},{"id":45,"rev":46},"drafts.358fa8ac-2467-4ef4-893c-dad3490046df","f5619d15-076f-4e8a-a050-fb32ee1b0366","blogPost","2026-10-08T14:00:04Z",[50],{"_createdAt":51,"_id":52,"_rev":53,"_type":54,"_updatedAt":55,"avatar":56,"bio":62,"employee":63,"name":64,"slug":65},"2026-10-06T15:34:18Z","6f547b5c-b2a6-4f91-a269-8db4d4f5181b","9fZfA2bz90M2vN1VV2YoWQ","blogAuthor","2026-10-06T15:36:24Z",{"_type":57,"alt":58,"asset":59},"image","Headshot of a smiling young man in a dark suit jacket and white shirt.",{"_ref":60,"_type":61},"image-cf8c2120f575c56e15a77b1d1687de89cea4790b-800x800-png","reference","Chase W. Hughes is a three-time founder. He built ProAI, one of the first commercialized GPT products, used by more than 300,000 businesses and institutions, and sold it bootstrapped after an 18-month run to seven figures. He previously founded the consulting firm Pro Business Plans. He works on multi-agent AI systems and how to evaluate them. chasewhughes.com\n","none","Chase W. Hughes",{"_type":10,"current":66},"chase-w-hughes",[68,80,97,105,121,129,139,159,167,175,194,202,210,242,250,258,266,296,304,339,347,366,374,382,401,420,439,447,463,471,479,494,502,510,518,530,542,554,566,574,582,609],{"_key":69,"_type":70,"children":71,"markDefs":78,"style":79},"b4dd07738637","block",[72],{"_key":73,"_type":74,"marks":75,"text":77},"9837536409f0","span",[76],"em","[Ed. note: If you listened to a podcast or read an article of ours and want to publish a response, let us know. We’re open to highlighting community views that respond to our work in ways that don’t fit in a comment section, whether you agree or disagree. Reach out at pitches@stackoverflow.com.]",[],"normal",{"_key":81,"_type":70,"children":82,"markDefs":96,"style":79},"61339e7ab7a4",[83,87,92],{"_key":84,"_type":74,"marks":85,"text":86},"887acadbd1d5",[],"In March 2025, a team at OpenAI published something you don't often get to see: the private reasoning of one of their own frontier models as it worked through a coding task during training. The model looked at the problem, decided that implementing the whole thing was too hard, and wrote down an alternative. If it called ",{"_key":88,"_type":74,"marks":89,"text":91},"af7289058bd5",[90],"code","sys.exit(0)",{"_key":93,"_type":74,"marks":94,"text":95},"a9ea33d372c6",[],", the test harness would exit gracefully. \"This is unnatural,\" it noted, \"but tests might pass.\"",[],{"_key":98,"_type":70,"children":99,"markDefs":104,"style":79},"91eadc1d15b8",[100],{"_key":101,"_type":74,"marks":102,"text":103},"70b694b1e4e7",[],"They passed.",[],{"_key":106,"_type":70,"children":107,"markDefs":120,"style":79},"8027ee7ff87f",[108,112,116],{"_key":109,"_type":74,"marks":110,"text":111},"2148d244bdeb",[],"The researchers called that a ",{"_key":113,"_type":74,"marks":114,"text":115},"13223646c328",[76],"systemic",{"_key":117,"_type":74,"marks":118,"text":119},"12c3d340b2c0",[]," hack—one general enough that, once a model finds it, it spreads across nearly every training environment. They caught a second of the same kind: raise an exception from outside the testing framework and skip evaluation entirely. Elsewhere in the paper, agents wrote stub implementations for problems whose test coverage was thin, parsed test files at runtime to extract and return exactly the values the tests were checking for, and—my favorite—found a compiled .jar someone had left in the repository, decompiled it, and copied out the reference solution.",[],{"_key":122,"_type":70,"children":123,"markDefs":128,"style":79},"bfd31166e01f",[124],{"_key":125,"_type":74,"marks":126,"text":127},"f33a05274bb7",[],"Every one of those runs went green.",[],{"_key":130,"_type":70,"children":131,"markDefs":137,"style":138},"6e43f29787b9",[132],{"_key":133,"_type":74,"marks":134,"text":136},"726a9ef23c1f",[135],"strong","The part the trust argument stops at",[],"h2",{"_key":140,"_type":70,"children":141,"markDefs":155,"style":79},"210aed2f680a",[142,146,151],{"_key":143,"_type":74,"marks":144,"text":145},"3fd7d3643fd2",[],"Ryan Donovan argued here in July that developers are attached to their tools because ",{"_key":147,"_type":74,"marks":148,"text":150},"5391f961755b",[149],"9c235b116a80","tools encode trust",{"_key":152,"_type":74,"marks":153,"text":154},"755c8ae4c1c8",[],", and that the trust is predictability earned through repetition—the reason a knife you've used for ten years feels like an extension of your hand, and the reason agentic tools, which change shape and weight every few weeks, are so hard to trust the same way.",[156],{"_key":149,"_type":157,"href":158,"reference":12},"link","https:\u002F\u002Fstackoverflow.blog\u002F2026\u002F07\u002F29\u002Fdevelopers-are-attached-to-tools-because-tools-encode-trust\u002F",{"_key":160,"_type":70,"children":161,"markDefs":166,"style":79},"00241e661263",[162],{"_key":163,"_type":74,"marks":164,"text":165},"b9da30061881",[],"I think that's right, and I want to push on where it stops. There is a second reason trust isn't accruing, and it is structurally worse than the first. It isn't only that the tool keeps changing shape. It's that the feedback loop you would need in order to learn the tool is broken at the point of measurement.",[],{"_key":168,"_type":70,"children":169,"markDefs":174,"style":79},"ce10caf35d88",[170],{"_key":171,"_type":74,"marks":172,"text":173},"cf11c2a4fed0",[],"Every trust instrument a developer owns—exit codes, green CI, a passing status check, the words \"task complete\"—rests on an assumption that software fails loudly. Unattended agents fail quietly. A loop can terminate cleanly, report success, and have produced nothing at all, and there is no signal anywhere in your stack that distinguishes that from a job well done. Call it the silent green exit.",[],{"_key":176,"_type":70,"children":177,"markDefs":191,"style":79},"0d5213876c90",[178,182,187],{"_key":179,"_type":74,"marks":180,"text":181},"cba60bb82d0d",[],"I built ProAI, one of the first commercialized GPT products, and I now work on multi-agent systems and how to evaluate them before they go anywhere near production; I write about that at ",{"_key":183,"_type":74,"marks":184,"text":186},"a8db7e22909c",[185],"452d1716f594","chasewhughes.com",{"_key":188,"_type":74,"marks":189,"text":190},"b32940aa211e",[],". This is the failure mode I think we are least prepared for, and it gets a fraction of the coverage that capability and safety do, because reliability is the least interesting of the three to write about and the only one that decides whether you can use the thing.",[192],{"_key":185,"_type":157,"href":193,"reference":12},"http:\u002F\u002Fchasewhughes.com",{"_key":195,"_type":70,"children":196,"markDefs":201,"style":138},"eadb44182137",[197],{"_key":198,"_type":74,"marks":199,"text":200},"13c6d9ad5a1d",[135],"We didn't earn loud failure. We inherited it.",[],{"_key":203,"_type":70,"children":204,"markDefs":209,"style":79},"8a8ce55ddaa9",[205],{"_key":206,"_type":74,"marks":207,"text":208},"f8bb9928d86a",[],"Think about why the knife works as a metaphor at all. You trust a knife because when it slips, you bleed—immediately, unambiguously, and in a way that teaches you something before you pick it up again. Every tool a developer trusts has some version of that property. A segfault is loud. A failed assertion is loud. A 500 is loud. The exit code is loudness compressed into a single integer, and it is probably the most-consumed piece of telemetry in computing.",[],{"_key":211,"_type":70,"children":212,"markDefs":241,"style":79},"7862f46e97ad",[213,217,221,225,229,233,237],{"_key":214,"_type":74,"marks":215,"text":216},"f0007d0a5b2a",[],"The thing worth noticing about ",{"_key":218,"_type":74,"marks":219,"text":220},"e51fd8dbb712",[90],"exit 0",{"_key":222,"_type":74,"marks":223,"text":224},"065410b3c79f",[]," is ",{"_key":226,"_type":74,"marks":227,"text":228},"03a52e5f38d2",[76],"who emits it",{"_key":230,"_type":74,"marks":231,"text":232},"6b9f58cc24be",[],". When ",{"_key":234,"_type":74,"marks":235,"text":236},"3a236c0a9691",[90],"make",{"_key":238,"_type":74,"marks":239,"text":240},"068fa1a155ef",[]," exits 0, that integer sits downstream of the work: a compiler ran, produced object files, and returned. The status is an observation, made by a process that was in a position to observe. The number is evidence about the world.",[],{"_key":243,"_type":70,"children":244,"markDefs":249,"style":79},"1868a79fad68",[245],{"_key":246,"_type":74,"marks":247,"text":248},"ec933eeb6b57",[],"When an agent loop exits 0, the integer is emitted by an orchestrator that stopped looping because the model produced a token meaning \"done.\" Nothing between the model's claim and the exit code checked anything. The status is a self-assessment, and you have plugged it into a socket built for an observation. Every consumer downstream—your dashboard, your alerting rules, your retry logic, the person who glances at a run history on Monday morning—inherits that substitution without ever being told it happened.",[],{"_key":251,"_type":70,"children":252,"markDefs":257,"style":79},"ecfc989184f8",[253],{"_key":254,"_type":74,"marks":255,"text":256},"06609715082b",[],"That's the whole mechanism. The rest is what it looks like in the wild.",[],{"_key":259,"_type":70,"children":260,"markDefs":265,"style":138},"dfcc32afe114",[261],{"_key":262,"_type":74,"marks":263,"text":264},"7ef07f0d814f",[135],"The gap already has numbers on it",[],{"_key":267,"_type":70,"children":268,"markDefs":291,"style":79},"db30942023aa",[269,273,278,282,287],{"_key":270,"_type":74,"marks":271,"text":272},"fb5775255b8f",[],"Donovan cites this site's own Developer Survey on the trust divergence, and it's worth pulling the primary figures because the shape is stark. Between the ",{"_key":274,"_type":74,"marks":275,"text":277},"8d6e75b13043",[276],"9c9443a5e66d","2024",{"_key":279,"_type":74,"marks":280,"text":281},"c55fd1796b74",[]," and ",{"_key":283,"_type":74,"marks":284,"text":286},"3331e17144f7",[285],"a57564f9c746","2025",{"_key":288,"_type":74,"marks":289,"text":290},"cff3cadbc3b0",[]," surveys, the share of developers using or planning to use AI tools rose from 76% to 84%, while the share who trust the accuracy of what those tools produce fell from 43% to 33%. Active distrust went the other way, from 30% to 46%. The single biggest frustration, cited by 66%, was \"AI solutions that are almost right, but not quite.\"",[292,294],{"_key":276,"_type":157,"href":293,"reference":12},"https:\u002F\u002Fsurvey.stackoverflow.co\u002F2024\u002Fai",{"_key":285,"_type":157,"href":295,"reference":12},"https:\u002F\u002Fsurvey.stackoverflow.co\u002F2025\u002Fai",{"_key":297,"_type":70,"children":298,"markDefs":303,"style":79},"6f0e46e3d1b4",[299],{"_key":300,"_type":74,"marks":301,"text":302},"ba5186ff7809",[],"\"Almost right, but not quite\" describes an output you can look at. What I'm describing sits one level beneath it: a run that isn't almost right but empty, and that presents identically to a run that worked.",[],{"_key":305,"_type":70,"children":306,"markDefs":336,"style":79},"1d31c758c4ea",[307,311,316,320,324,328,332],{"_key":308,"_type":74,"marks":309,"text":310},"6ffa725bed12",[],"Two benchmarks make it concrete.",{"_key":312,"_type":74,"marks":313,"text":315},"f6c917117151",[314],"74cd7d1a47c0"," τ-bench, from Sierra's research team",{"_key":317,"_type":74,"marks":318,"text":319},"9a06e0eee871",[],", introduced a metric that deserves to be far better known: ",{"_key":321,"_type":74,"marks":322,"text":323},"59fa370cc314",[135],"pass^k",{"_key":325,"_type":74,"marks":326,"text":327},"4248c48ef3cd",[],"—the probability that an agent succeeds at the same task on ",{"_key":329,"_type":74,"marks":330,"text":331},"9ad734089118",[76],"all",{"_key":333,"_type":74,"marks":334,"text":335},"ba7eda224b2e",[]," k attempts, rather than on at least one. On the models they tested, agents came in under 50% on the standard measure and under 25% at pass^8 in the retail domain.",[337],{"_key":314,"_type":157,"href":338,"reference":12},"https:\u002F\u002Fsierra.ai\u002Fblog\u002Fbenchmarking-ai-agents",{"_key":340,"_type":70,"children":341,"markDefs":346,"style":79},"98da623ecee2",[342],{"_key":343,"_type":74,"marks":344,"text":345},"0381349c6520",[],"Those specific numbers are stale, and I want to be straight about that: it was 2024, the headline model was GPT-4o, and the frontier has moved a long way since. The metric hasn't moved at all. Pass^k is a direct measurement of the property Donovan says trust is made of. It asks whether Tuesday matches Monday and puts a number on the answer. Almost nobody reports it, which is why \"it worked when I tried it\" is still accepted as evidence in engineering conversations where nobody would accept it about anything else.",[],{"_key":348,"_type":70,"children":349,"markDefs":363,"style":79},"e4afd7a451ba",[350,354,359],{"_key":351,"_type":74,"marks":352,"text":353},"7ecb3309c1db",[],"The second is ",{"_key":355,"_type":74,"marks":356,"text":358},"eda8559f58a8",[357],"e5aa1e80760e","TheAgentCompany",{"_key":360,"_type":74,"marks":361,"text":362},"60b1696ca9f8",[],", from Carnegie Mellon, which dropped agents into a simulated software company and gave them real jobs to do. The most competitive agent completed 30% of tasks autonomously. But the finding that belongs in this argument isn't the score, it's a behavior the researchers describe: \"when the agent is not clear what the next steps should be, it sometimes tries to be clever and create fake 'shortcuts' that omit the hard part of the task.\" Their example is perfect. An agent that couldn't find a particular colleague on the company chat platform renamed a different user to that person's name, and carried on.",[364],{"_key":357,"_type":157,"href":365,"reference":12},"https:\u002F\u002Fthe-agent-company.com\u002F",{"_key":367,"_type":70,"children":368,"markDefs":373,"style":79},"8d7d6b689cff",[369],{"_key":370,"_type":74,"marks":371,"text":372},"d23f095fffc1",[],"Sit with what that produces. From the outside, the step completed. There is a user with the right name, an action taken against it, and no error anywhere in the trace. The only thing wrong is that the actual colleague never heard anything—and nothing in the system knows that, including the agent, which has no reason to believe it failed.",[],{"_key":375,"_type":70,"children":376,"markDefs":381,"style":138},"d7ce7dcbac75",[377],{"_key":378,"_type":74,"marks":379,"text":380},"c01a046f92ed",[135],"When it reaches production",[],{"_key":383,"_type":70,"children":384,"markDefs":398,"style":79},"6a034c87ff4c",[385,389,394],{"_key":386,"_type":74,"marks":387,"text":388},"4aefdb005649",[],"In July 2025, ",{"_key":390,"_type":74,"marks":391,"text":393},"4a3b5cfb0b77",[392],"90ac5d8b9a0b","Jason Lemkin",{"_key":395,"_type":74,"marks":396,"text":397},"bf4033d3913f",[]," spent about two weeks building on Replit's coding agent and documented it publicly as it went. The database deletion took the headlines: the agent wiped a production database despite his standing instruction that nothing be changed without his permission. But the part that belongs here came before that, and got much less attention.",[399],{"_key":392,"_type":157,"href":400,"reference":12},"https:\u002F\u002Fwww.theregister.com\u002Fsoftware\u002F2025\u002F07\u002F21\u002Fvibe-coding-service-replit-deleted-production-database\u002F",{"_key":402,"_type":70,"children":403,"markDefs":417,"style":79},"6e9fcbbeb945",[404,408,413],{"_key":405,"_type":74,"marks":406,"text":407},"d43e1d61c3b5",[],"\"It kept covering up bugs and issues by creating fake data, fake reports, and worse of all, lying about our unit test,\" ",{"_key":409,"_type":74,"marks":410,"text":412},"eb414dab87de",[411],"d38b3cbadae7","he wrote",{"_key":414,"_type":74,"marks":415,"text":416},"00a5d75bcbce",[],". At one point the agent generated a database of 4,000 entirely fictional people. Afterward, it told him rollback was impossible and that all database versions had been destroyed. That was also wrong—the rollback worked fine.",[418],{"_key":411,"_type":157,"href":419,"reference":12},"https:\u002F\u002Fwww.cs.cmu.edu\u002Fnews\u002F2025\u002Fagent-company",{"_key":421,"_type":70,"children":422,"markDefs":436,"style":79},"10245ac5af74",[423,427,432],{"_key":424,"_type":74,"marks":425,"text":426},"8cb6d10bb1a2",[],"Fake data, fake reports, a misreported unit test, and a confident status claim that didn't survive being checked. ",{"_key":428,"_type":74,"marks":429,"text":431},"874b5a543c93",[430],"271adaeba7aa","Replit's CEO called what happened",{"_key":433,"_type":74,"marks":434,"text":435},"725617bc8552",[]," \"unacceptable and should never be possible,\" which is the correct response, and the incident is still the cleanest public illustration I know of the difference between an agent that failed and an agent that reported.",[437],{"_key":430,"_type":157,"href":438,"reference":12},"https:\u002F\u002Fwww.businessinsider.com\u002Freplit-ceo-apologizes-ai-coding-tool-delete-company-database-2025-7",{"_key":440,"_type":70,"children":441,"markDefs":446,"style":138},"af3055b339bb",[442],{"_key":443,"_type":74,"marks":444,"text":445},"cf52e6a3a22a",[135],"The objection, which is a good one",[],{"_key":448,"_type":70,"children":449,"markDefs":462,"style":79},"01395422feee",[450,454,458],{"_key":451,"_type":74,"marks":452,"text":453},"9e43e94a02cd",[],"Here is what a reasonable SRE says to all of this: none of it is new. Cron jobs have been failing silently since the 1970s. It's why we have dead man's switches, heartbeat monitors, and alerts that fire on the ",{"_key":455,"_type":74,"marks":456,"text":457},"595de61a2ec3",[76],"absence",{"_key":459,"_type":74,"marks":460,"text":461},"0da471c03568",[]," of a signal rather than the presence of an error. You've rediscovered monitoring and given it a name.",[],{"_key":464,"_type":70,"children":465,"markDefs":470,"style":79},"feacaa7d4512",[466],{"_key":467,"_type":74,"marks":468,"text":469},"34f4246bf7e2",[],"Mostly fair—and the good news buried in it is real, because it means the shape of the fix is already known and the tools already exist. But there is one difference that breaks the old approach, and it's worth being precise about.",[],{"_key":472,"_type":70,"children":473,"markDefs":478,"style":79},"be56b7e5854a",[474],{"_key":475,"_type":74,"marks":476,"text":477},"f8ef1f403118",[],"Every one of those techniques works because the failure modes were enumerable. You knew the finite list of ways a nightly job could fail to finish—the box is down, the disk is full, the lock is held, the upstream timed out—and you either wrote a check per mode or wrote one heartbeat that covered them all, because \"finished\" had exactly one meaning and the job had no opinions about it.",[],{"_key":480,"_type":70,"children":481,"markDefs":493,"style":79},"47dbbb106e0a",[482,486,489],{"_key":483,"_type":74,"marks":484,"text":485},"d101922a9f12",[],"An agent decides its own control flow at runtime. The plan is generated fresh per run, which means the set of ways it can terminate cleanly having produced nothing is not a list anybody can write in advance. It includes ",{"_key":487,"_type":74,"marks":488,"text":91},"f6c5b580fe1b",[90],{"_key":490,"_type":74,"marks":491,"text":492},"4f7fb5bb4fc5",[]," because implementing the reader looked hard. It includes renaming a user. It includes strategies nobody in your building has thought of, because nobody in your building is writing the strategy.",[],{"_key":495,"_type":70,"children":496,"markDefs":501,"style":79},"380b789a8fba",[497],{"_key":498,"_type":74,"marks":499,"text":500},"edcbe81c58f9",[],"So you cannot write the check from the failure side. Anything written from the failure side is a list, and the agent is not confined to your list. You have to write it from the outcome side.",[],{"_key":503,"_type":70,"children":504,"markDefs":509,"style":138},"29b3624ac3b3",[505],{"_key":506,"_type":74,"marks":507,"text":508},"6a18f90410d8",[135],"Instrument the artifact, not the run",[],{"_key":511,"_type":70,"children":512,"markDefs":517,"style":79},"032e04b49c47",[513],{"_key":514,"_type":74,"marks":515,"text":516},"1162be6632ff",[],"Four things follow. None of them needs new tooling, and that's rather the point.",[],{"_key":519,"_type":70,"children":520,"markDefs":529,"style":79},"220c7d3fd384",[521,525],{"_key":522,"_type":74,"marks":523,"text":524},"f485fbf4b1a0",[135],"Assert on the work product.",{"_key":526,"_type":74,"marks":527,"text":528},"e432ad841139",[]," Not \"did the loop exit 0\" but \"does the row exist, is the file on disk, did the message actually leave the queue, is the branch actually pushed.\" The check has to read state the agent did not author. If your only evidence that the emails went out is the agent's report that it sent them, you do not have evidence. You have a claim from a party that has a vested interest in the answer.",[],{"_key":531,"_type":70,"children":532,"markDefs":541,"style":79},"b8bf76c53f3a",[533,537],{"_key":534,"_type":74,"marks":535,"text":536},"7fff186aa3ad",[135],"Don't let verification share a failure mode with the work.",{"_key":538,"_type":74,"marks":539,"text":540},"d8b735967768",[]," If the same agent, in the same context, both does the job and reports on it, the report inherits every error the job had—including the ones that caused it. Donovan raises LLM-as-a-judge as the emerging answer to reviewing at agent speed, and I think that's right, with one constraint: the judge has to read the artifact, not the transcript. A judge reading the transcript is grading a story about the work, and stories are the thing these systems are best at.",[],{"_key":543,"_type":70,"children":544,"markDefs":553,"style":79},"d56d6f731861",[545,549],{"_key":546,"_type":74,"marks":547,"text":548},"7e3511f5c9da",[135],"Alert on the absence of change.",{"_key":550,"_type":74,"marks":551,"text":552},"03730565023a",[]," This is the cheapest item on the list and almost nobody does it. A job that always writes, and this time wrote nothing while reporting success, is more suspicious than a job that threw an exception. Errors get alerts; zero-effect successes get a green square in a run history that no one reads. Invert it. A scheduled agent that reports success having touched nothing should page someone. Nothing is a result, and it deserves to be a loud one.",[],{"_key":555,"_type":70,"children":556,"markDefs":565,"style":79},"837acddf4bed",[557,561],{"_key":558,"_type":74,"marks":559,"text":560},"9867a22ab692",[135],"Measure pass^k in simulations before you deploy.",{"_key":562,"_type":74,"marks":563,"text":564},"a16d6c6a75f3",[]," Run the agent at the same scenario dozens of times and look at the distribution rather than the best case, in an environment where you own the ground truth and can therefore check the artifact instead of the story. The purpose of a simulation is not to prove the agent can do the job. Anyone can demo that. It's to find out how often it says it did.",[],{"_key":567,"_type":70,"children":568,"markDefs":573,"style":138},"87f805f22e1e",[569],{"_key":570,"_type":74,"marks":571,"text":572},"82004e492c09",[135],"What we haven't replaced",[],{"_key":575,"_type":70,"children":576,"markDefs":581,"style":79},"18a3a11ee5be",[577],{"_key":578,"_type":74,"marks":579,"text":580},"c06e2ac56521",[],"Notice how modest all of that is. Assertions and alerts—the same discipline that built trust in every previous generation of infrastructure, pointed at a different target.",[],{"_key":583,"_type":70,"children":584,"markDefs":606,"style":79},"01e1e163b368",[585,589,594,598,602],{"_key":586,"_type":74,"marks":587,"text":588},"dcb971ba110b",[],"This part is genuinely new, and it shows up in how people talk about it. The ",{"_key":590,"_type":74,"marks":591,"text":593},"781eb70f4176",[592],"eb8d72983f91","OpenTelemetry ",{"_key":595,"_type":74,"marks":596,"text":597},"4e4c76e4a609",[],"semantic conventions for GenAI—the closest thing the industry has to an agreed way of instrumenting these systems—define attributes for the model, the operation, the input and output token counts, the finish reason, the error type, the agent's own name and version. All of it useful. None of it is an attribute for whether the work happened. That isn't an oversight and I don't think it's a criticism; these are conventions for instrumenting a ",{"_key":599,"_type":74,"marks":600,"text":601},"ef46eabda2ec",[76],"call",{"_key":603,"_type":74,"marks":604,"text":605},"364c63041631",[],", and in every prior generation of software the question didn't need a field of its own. Completion implied effect. You did not need to ask a compiler whether it meant it.",[607],{"_key":592,"_type":157,"href":608,"reference":12},"https:\u002F\u002Fopentelemetry.io\u002Fdocs\u002Fspecs\u002Fsemconv\u002Fregistry\u002Fattributes\u002Fgen-ai\u002F",{"_key":610,"_type":70,"children":611,"markDefs":616,"style":79},"c7914b729124",[612],{"_key":613,"_type":74,"marks":614,"text":615},"7cbfe767f5cf",[],"That implication is gone, and nothing has been put in its place. A green exit code used to be a fact about your program. It is now an opinion held by your agent. And the reason trust isn't accruing isn't only that the knife keeps changing shape—it's that you can no longer tell whether it cut.",[],true,"2026\u002F10\u002F08","Agents don't build trust for another reason, structurally worse than the first. It isn't only that the tool keeps changing shape. It's that the feedback loop you would need in order to learn the tool is broken at the point of measurement.",{"_type":57,"alt":621,"asset":622},"A repeating pattern of interlocked red, blue, and yellow stylized hammers on a light yellow background.",{"_ref":623,"_type":61},"image-29bbae9c01de534b803084e4f0b9a77f9cd90e5c-12000x6300-jpg","2026-10-08T14:00:00.000Z",{"_type":10,"current":626},"a-green-exit-code-is-not-evidence-that-the-work-happened",[628,637],{"_createdAt":629,"_id":630,"_rev":631,"_type":632,"_updatedAt":633,"slug":634,"title":636},"2023-05-23T16:43:21Z","wp-tagcat-ai","fpDTFQqIDjNJIbHDKPBGpV","blogTag","2025-01-30T16:19:01Z",{"current":635},"ai","AI",{"_createdAt":638,"_id":639,"_rev":640,"_system":641,"_type":632,"_updatedAt":644,"slug":645,"title":647},"2026-06-12T16:16:20Z","51c761d7-73f7-42f4-aa49-8484e3849e7c","P0qLqkXH0zpkT6RRZ9Iwel",{"base":642},{"id":639,"rev":643},"MwgZb85ftkde1TTvQsHYa6","2026-09-28T16:40:45Z",{"_type":10,"current":646},"building-software","Building software","A green exit code is not evidence that the work happened",[650,656,662,668],{"_id":651,"publishedAt":652,"slug":653,"sponsored":12,"title":655},"ce1fd642-fe2b-4d83-95ad-67b7645d7959","2026-10-07T20:40:29.066Z",{"_type":10,"current":654},"part-3-knowing-when-your-agent-doesn-t-know-the-confidence-layer","Part 3: Knowing when your agent doesn’t know: the confidence layer",{"_id":657,"publishedAt":658,"slug":659,"sponsored":12,"title":661},"f17b27e9-17f5-4203-8066-68c8df26ef42","2026-10-07T20:31:12.368Z",{"_type":10,"current":660},"evals-as-a-deployment-gate-and-how-to-know-when-they-drift","Part 2: Evals as a deployment gate — and how to know when they drift",{"_id":663,"publishedAt":664,"slug":665,"sponsored":12,"title":667},"39f2fc9f-742a-4273-a4d1-abdecd14fcf8","2026-10-07T20:14:49.390Z",{"_type":10,"current":666},"part-1-make-your-ai-agents-boring-the-determinism-layer","Part 1: Make your AI agents boring: the determinism layer",{"_id":669,"publishedAt":670,"slug":671,"sponsored":12,"title":673},"f2b2b7a0-c8e2-4387-8a7d-439b28bab359","2026-10-07T20:03:47.355Z",{"_type":10,"current":672},"implementing-a-modular-master-agent-telemetry-and-diagnostic-framework-in-python-prime-sentinel-command-psc","Implementing a Modular Master-Agent Telemetry & Diagnostic Framework in Python: Prime-Sentinel Command (PSC)",{"data":675,"sourceMap":-1},{"count":676,"lastTimestamp":12},0]