<?xml version="1.0" encoding="UTF-8"?><rss xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:atom="http://www.w3.org/2005/Atom" version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:googleplay="http://www.google.com/schemas/play-podcasts/1.0"><channel><title><![CDATA[Nikolaos Vasiloglou]]></title><description><![CDATA[Researcher]]></description><link>https://nikolaosvasiloglou.substack.com</link><image><url>https://substackcdn.com/image/fetch/$s_!Ll0-!,w_256,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fe984a3e7-f1db-4170-b916-a169a59a4550_261x261.jpeg</url><title>Nikolaos Vasiloglou</title><link>https://nikolaosvasiloglou.substack.com</link></image><generator>Substack</generator><lastBuildDate>Wed, 19 Aug 2026 20:45:46 GMT</lastBuildDate><atom:link href="https://nikolaosvasiloglou.substack.com/feed" rel="self" type="application/rss+xml"/><copyright><![CDATA[Nikolaos Vasiloglou]]></copyright><language><![CDATA[en]]></language><webMaster><![CDATA[nikolaosvasiloglou@substack.com]]></webMaster><itunes:owner><itunes:email><![CDATA[nikolaosvasiloglou@substack.com]]></itunes:email><itunes:name><![CDATA[Nikolaos Vasiloglou]]></itunes:name></itunes:owner><itunes:author><![CDATA[Nikolaos Vasiloglou]]></itunes:author><googleplay:owner><![CDATA[nikolaosvasiloglou@substack.com]]></googleplay:owner><googleplay:email><![CDATA[nikolaosvasiloglou@substack.com]]></googleplay:email><googleplay:author><![CDATA[Nikolaos Vasiloglou]]></googleplay:author><itunes:block><![CDATA[Yes]]></itunes:block><item><title><![CDATA[How This Analysis Was Built: Conference Knowledge Extraction as an Agentic Pipeline]]></title><description><![CDATA[I did not build the ICLR 2026 analysis by sitting down with 9,000 papers and reading them like a normal person.]]></description><link>https://nikolaosvasiloglou.substack.com/p/how-this-analysis-was-built-conference</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/how-this-analysis-was-built-conference</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Fri, 14 Aug 2026 12:11:04 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!0JJK!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!0JJK!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!0JJK!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!0JJK!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:1312953,&quot;alt&quot;:&quot;A sparse cartoon pipeline shows papers entering a funnel, becoming a source map, passing through an agent workbench, and emerging as an inspected article.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A sparse cartoon pipeline shows papers entering a funnel, becoming a source map, passing through an agent workbench, and emerging as an inspected article." title="A sparse cartoon pipeline shows papers entering a funnel, becoming a source map, passing through an agent workbench, and emerging as an inspected article." srcset="https://substackcdn.com/image/fetch/$s_!0JJK!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!0JJK!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F67d348d7-323c-4a9c-8b74-ac5c2e30682f_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Papers move through ingestion, a source map, an agent workbench, and human article review.</figcaption></figure></div><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!E3o8!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!E3o8!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!E3o8!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/b8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:149253,&quot;alt&quot;:&quot;A paper-boat navigator works at a laptop along a red dashed route between islands for paper ingestion, a source map with linked pins, the coding agent, and a finished article draft with a flag.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator works at a laptop along a red dashed route between islands for paper ingestion, a source map with linked pins, the coding agent, and a finished article draft with a flag." title="A paper-boat navigator works at a laptop along a red dashed route between islands for paper ingestion, a source map with linked pins, the coding agent, and a finished article draft with a flag." srcset="https://substackcdn.com/image/fetch/$s_!E3o8!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!E3o8!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fb8075e56-3158-4b39-a307-d4a723d7d9dc_3200x1800.png 1456w" sizes="100vw"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">The pipeline moves from paper ingestion through a source map and coding agent to an inspectable article draft.</figcaption></figure></div><p>I did not build the ICLR 2026 analysis by sitting down with 9,000 papers and reading them like a normal person. That would have been noble, impossible, and not very useful. The point of this project was to test whether a conference at this scale can be turned into usable executive knowledge through an agentic pipeline: scraping, parsing, topic clustering, local model analysis, NotebookLM synthesis, Claude Code and Codex automation, knowledge-graph extraction, agentic orchestration, and then human editorial judgment at the end.</p><h2>The Short Version</h2><ul><li><p><strong>The pipeline has four layers:</strong> ingestion and normalization, bulk analysis, narrative synthesis, and human editorial judgment. Each one can fail in a way the next cannot repair.</p></li><li><p><strong>Roughly 80 to 90 percent of the volume work ran locally</strong>, on my laptops, with Qwen and other smaller models. A frontier-only workflow would have been too expensive and too slow to iterate at this scale.</p></li><li><p><strong>The remaining 10 to 20 percent still needed cloud models or human supervision.</strong> This was not a fully autonomous pipeline, and claiming otherwise would make it less credible.</p></li><li><p><strong>The output is not just prose.</strong> It is source maps, evidence buckets, and thesis memos &#8212; structured enough to interrogate, and to reuse.</p></li><li><p><strong>The honest caveat:</strong> the method is fragile and tool-dependent. NotebookLM needs steering, transcripts need QA, local models need verification, and topic clusters need human curation.</p></li></ul><h2>Ingestion: The Unglamorous Foundation</h2><p>The pipeline began with ingestion. Conference material had to be collected, converted, and normalized: abstracts, paper text where available, topic reports, transcripts, slides, audio-derived podcast material, and analysis reports. That work is not glamorous, but it is the foundation. A language model cannot reason over a conference if the source material is scattered, mislabeled, or trapped in formats it cannot reliably consume.</p><p>The transcripts in particular needed quality control. Some original OpenAI podcast transcripts were truncated or suspicious, so five topics were rerun with chunk validation. The local-model transcripts were useful as cross-checks, but the source authority map marks which transcript is authoritative for each topic. This is the kind of work a polished summary hides. It should not be hidden. If the input transcript is incomplete, the resulting claim can be wrong before the model starts reasoning.</p><h2>Analysis: Local Models Did the Heavy Lifting</h2><p>The next layer was analysis. I used a mix of local models and coding agents to organize the corpus into topic areas, generate reports, compare evidence, and produce summaries. Roughly 80 to 90 percent of this volume work ran locally on my laptops - Beauty and the Beasts - with Qwen and other smaller models doing the heavy repetitive passes. That mattered because a frontier-only workflow would have been too expensive and too slow to iterate at this scale.</p><p>The cloud models and coding agents were still necessary. Claude Code and Codex helped automate file processing, registry work, markdown conversion, source mapping, and planning. Frontier models were useful for hard synthesis and review. The honest split is that local systems handled the bulk volume, while 10 to 20 percent still required stronger cloud compute or more deliberate human supervision. This was not a fully autonomous pipeline, and claiming otherwise would make it less credible.</p><h2>Synthesis: NotebookLM Was Useful, Not an Oracle</h2><p>NotebookLM played a different role. It was useful for narrative synthesis: podcasts, topic walkthroughs, and repeated listening. It helped me hear the conference back to myself. But it was not the oracle. NotebookLM can miss important points, choose the wrong emphasis, and produce titles that skew pessimistic or clickbait. I changed titles because the generated versions often reached for drama instead of signal. That editorial override is not decoration; it is part of the filter.</p><h2>What the Pipeline Produced</h2><p>It produced executive summaries, topic reports, source maps, thesis memos, series architecture, and the six posts in this series. More importantly, it produced a way to interrogate the conference. I could ask which claims were first-party thesis, which were conference evidence, which came from podcast synthesis, and which were public company claims. That separation is the difference between "a model summarized ICLR" and "a source-grounded analysis exists."</p><p>The process still has no perfect name. "Conference distillation" is close. "Machine-readable knowledge extraction" is closer. The important distinction is that the output is not only for human reading. The goal is to create structured material that can be reused: by people writing essays, by agents building software, by knowledge graphs, by future conference comparisons, and by systems that need to reason over the research record.</p><h2>The Caveat, and Why the Method Still Matters</h2><p>That ambition creates a danger: overclaiming. I have not solved machine-readable knowledge extraction. The pipeline is fragile, tool-dependent, and full of editorial judgment. NotebookLM needs steering. Transcripts need QA. Local models need verification. Cloud models can still hallucinate. Topic clusters need human curation. The series titles and thesis architecture were not outsourced to a podcast generator because the whole point is to preserve judgment, not erase it.</p><p>Still, the method matters. A conference of this size is no longer readable by ordinary human attention alone. If we want executives, researchers, and builders to make decisions from the research frontier, we need pipelines that turn giant corpora into inspectable claims. That means not only summarizing papers, but preserving source authority, tracking uncertainty, separating evidence buckets, and leaving enough structure for later reuse.</p><p>The best way to read this series is therefore with two layers in mind. The first layer is the argument about ICLR 2026: filtering, data curation, verification, local models, and process-grounded AI. The second layer is the method that made the argument possible. If the method feels a little unfinished, that is because the field is unfinished. We are learning how to read conferences with machines while still keeping humans responsible for what the reading means.</p>]]></content:encoded></item><item><title><![CDATA[The Training Data Business: Creating a Moat in AI]]></title><description><![CDATA[The previous posts leave us with a practical business question.]]></description><link>https://nikolaosvasiloglou.substack.com/p/the-training-data-business-creating</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/the-training-data-business-creating</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Fri, 14 Aug 2026 12:09:29 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!lWG7!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!lWG7!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!lWG7!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!lWG7!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:1373340,&quot;alt&quot;:&quot;A paper-boat navigator turns a dial connected to a highlighted middle band of increasingly complex task graphs.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator turns a dial connected to a highlighted middle band of increasingly complex task graphs." title="A paper-boat navigator turns a dial connected to a highlighted middle band of increasingly complex task graphs." srcset="https://substackcdn.com/image/fetch/$s_!lWG7!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!lWG7!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F8e3f19f7-f74f-4be6-a489-2b1664ac5476_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">A complexity dial selects the narrow band of tasks just beyond current capability.</figcaption></figure></div><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!4nZz!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!4nZz!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!4nZz!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:158980,&quot;alt&quot;:&quot;A paper-boat navigator sails a red dashed route past four islands: a page with X marks for model failures, a highlighted wave band for the just-missed difficulty band, a gauge for the complexity dial, and a stack of checked cards for verified tasks.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator sails a red dashed route past four islands: a page with X marks for model failures, a highlighted wave band for the just-missed difficulty band, a gauge for the complexity dial, and a stack of checked cards for verified tasks." title="A paper-boat navigator sails a red dashed route past four islands: a page with X marks for model failures, a highlighted wave band for the just-missed difficulty band, a gauge for the complexity dial, and a stack of checked cards for verified tasks." srcset="https://substackcdn.com/image/fetch/$s_!4nZz!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!4nZz!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F94c36489-545b-47a7-bb6e-779331ff2543_3200x1800.png 1456w" sizes="100vw"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">The most useful training tasks sit just beyond current capability and can be tuned with an ontology complexity dial.</figcaption></figure></div><p>The previous posts leave us with a practical business question. If cheap passive data is becoming less sufficient, and if agent learning increasingly depends on executable worlds, then who is going to manufacture the practice?</p><p>That is the training data business. Not the old business of scraping more text, labeling more examples, or buying a larger corpus. The new business is creating verified tasks in worlds where an agent can act, observe, fail, recover, and learn. The scarce object is not a document. It is a well-formed challenge.</p><h2>The Short Version</h2><p>If you only read one section, read this one.</p><ul><li><p><strong>Most tasks are worthless for training.</strong> A task the model already solves is regression data. A task it cannot approach gives almost no signal, because pass/fail on something never passed is empty. The valuable task is the one the model <em>just misses</em>.</p></li><li><p><strong>That "just-missed band" is the real bottleneck.</strong> ICLR 2026 work says it directly: RLVR depends on well-crafted tasks with ground truth, and the difficulty of generated agentic tasks is hard to control. The frontier model may be expensive, but the frontier task <em>generator</em> is scarce.</p></li><li><p><strong>Owning a real environment is not enough &#8212; you need a difficulty dial.</strong> If a task is too easy, can you lengthen the reasoning path without changing the domain? Too hard, can you remove exactly one hidden dependency? Can you tell a missing fact from a bad tool call from a misread business rule?</p></li><li><p><strong>Ontology is a complexity dial.</strong> That is our synthesis. Once entities, relationships, constraints, permissions, state transitions, and exceptions are explicit, task generation becomes controlled perturbation of a formal world rather than prompt writing &#8212; and the same structure lets you verify final state, not just output text.</p></li><li><p><strong>The moat is not the PDF, the meeting note, or the raw log.</strong> It is the ability to turn domain structure into a stream of verified, calibrated tasks &#8212; and to know when a model is exactly one abstraction away from learning.</p></li></ul><h2>What's in This Post</h2><p>Each section opens with the short version, then a deeper dive with the paper-level evidence. Read the short versions for the argument; stay for the deep dives when you want the receipts.</p><ol><li><p><strong>The Just-Missed Band</strong> &#8212; why most tasks produce no learning signal.</p></li><li><p><strong>Environments Need a Difficulty Dial</strong> &#8212; where "environment" becomes an engineering discipline.</p></li><li><p><strong>Ontology Is a Complexity Dial</strong> &#8212; how to make task difficulty adjustable, and verifiable.</p></li><li><p><strong>What This Changes on Your Roadmap</strong> &#8212; the questions to ask instead of "what data do we have?"</p></li></ol><h2>The Just-Missed Band</h2><p><strong>The short version:</strong> For reinforcement learning, not every challenge is useful. A task the model already solves is mostly regression data &#8212; it confirms yesterday's model still works. A task the model cannot approach is also weak: if the only reward is pass or fail and the model never gets close, the learning signal is almost empty. The valuable task is the one the model just misses: hard enough to expose a real capability gap, structured enough that the model can make partial progress, trigger useful feedback, receive partial credit, and try again.</p><h3>Deeper dive: the conference evidence that this band is scarce</h3><p>You can sometimes brute-force your way past a too-hard task with enormous sampling budgets, but that is not a business model most companies can afford.</p><p>Several ICLR 2026 papers suggest that this band is becoming a bottleneck. Don't Just Fine-tune the Agent, Tune the Environment<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a> starts from the problem directly: LLM-agent development is hampered by scarce high-quality training data, synthetic SFT data can overfit, and standard RL struggles with cold start and instability in complex multi-turn tool use. Its answer is not simply "more data." It is Environment Tuning<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a>: a structured curriculum, environment augmentation, and fine-grained progress rewards.</p><p>Supervised Reinforcement Learning<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a> makes the same point from another angle. RL with verifiable rewards can fail when correct solutions are rarely sampled even after many attempts. The paper's response is smoother step-wise supervision, giving the model richer signals even when all rollouts are wrong. In other words, "wrong" is too coarse a category. The training system needs to know how the model was wrong.</p><p>Search Self-Play<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a> puts the business problem in one sentence: RLVR depends on well-crafted task queries and ground-truth answers, while the difficulty of generated agentic tasks is hard to control. That is the strategic bottleneck. The frontier model may be expensive, but the frontier task generator is scarce.</p><h2>Environments Need a Difficulty Dial</h2><p><strong>The short version:</strong> Real environments help, but they do not solve the problem by themselves. You can buy real data, collect logs, or build a cloud lab. But once you have it: what is the difficulty dial? If the task is too easy, can you lengthen the reasoning path without changing the domain? If it is too hard, can you remove exactly one hidden dependency? If the agent fails, can you tell whether the failure came from a missing fact, a wrong traversal, a bad tool call, a state update error, or a misunderstanding of the business rule? This is where "environment" stops being a generic word and becomes an engineering discipline.</p><h3>Deeper dive: four dials from ICLR 2026</h3><p>AgentGym-RL<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a> treats agent training as long-horizon decision making across realistic environments. Its staged interaction approach begins with short-horizon interaction and progressively expands the interaction length so agents can stabilize before deeper exploration. That is a complexity dial over time and horizon.</p><p>CodeGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> transforms static coding problems into interactive, verifiable, controllable multi-turn tool-use environments. That is a complexity dial over tools, executable steps, and workflow shape.</p><p>BIRD-INTERACT<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> shows the same idea in database work. It couples databases with a hierarchical knowledge base, metadata, a user simulator, CRUD tasks, execution feedback, and executable tests. That is much closer to enterprise reality than a single text-to-SQL prompt. It is also a reminder that the world is not just the database. The world is the database plus the business semantics plus the user plus the allowed actions plus the success condition.</p><p>OpenApps<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a> takes the UI-agent version of the same problem. Instead of testing an agent on one fixed clone of an app, it generates thousands of configurable app variations across messenger, calendar, maps, and related interfaces. Reliability changes dramatically across those variations. That matters because a fixed environment can hide fragility. A configurable environment can reveal it.</p><p>These papers point toward a different view of data. Data quality is not only whether the tokens are clean. It is whether the task produces useful learning pressure. Scaling Laws Revisited<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a> makes the broader scaling point explicit by adding a data-quality parameter to the Chinchilla-style view of model size and dataset volume. Why Less is More (Sometimes)<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a> studies curation by difficulty and correctness, showing conditions where smaller curated datasets can outperform full datasets. The point is not that scale stopped mattering. The point is that scale without quality, difficulty, and verifier design is an incomplete strategy.</p><h2>Ontology Is a Complexity Dial</h2><p><strong>The short version:</strong> That is our synthesis. An ontology is not the only way to model a world, but for enterprise worlds it is one of the cleanest ways to make task difficulty adjustable. A semantic model gives you entities, relationships, constraints, permissions, state transitions, exceptions, and business rules. Once those are explicit, task generation becomes less like prompt writing and more like controlled perturbation of a formal world &#8212; and the same structure lets the training system grade final state against the rules of the environment, not just check whether an output string looks right.</p><h3>Deeper dive: turning the dial in both directions</h3><p>You can make a task easier by shortening a traversal, removing an exception, reducing the number of entities, revealing an intermediate state, narrowing the action set, or making the success condition more local. You can make it harder by adding a dependency, introducing a conflicting constraint, requiring reconciliation across two systems, hiding a needed fact behind a tool call, adding temporal state, or asking for a final action that must satisfy several business rules at once.</p><p>That is the difference between a static collection of examples and a training-data machine. A static collection has fixed coverage; a machine can generate another task at the right level.</p><p>The ontology also helps with the verification problem raised in the previous post. If the world is formal enough, the training system can check not just whether an output string looks right, but whether the final state satisfies the rules of the environment. In principle, when connected to executable tests or state checks, it can grade intermediate steps, expose partial progress, and distinguish a semantic failure from a formatting failure. The richer the world model, the more precise the feedback can become.</p><h2>What This Changes on Your Roadmap</h2><p><strong>The short version:</strong> At first glance, "making tasks" sounds like routine production work. In reality it is the work of encoding a domain so that an agent can practice inside it. Most companies do not need to own generic generation. They need to own the worlds where generic models become useful. So the roadmap question shifts: not "what data do we have?" but "what world can we simulate, what actions are allowed, what state changes, what does success mean, and how do we generate the next task the model just misses?"</p><h3>Deeper dive: what the moat actually is, with an example</h3><p>The moat is not the PDF, the meeting note, or the raw log. The moat is the ability to turn domain structure into a stream of verified, calibrated tasks. It is the ability to know when a model already understands a workflow, when it is outside the relevant capability range, and when it is exactly one abstraction away from learning.</p><p>A tax business setup makes the pattern concrete. Taxes are full of entities, rules, exceptions, thresholds, documents, jurisdictions, and stateful decisions. That makes them painful for static prompting and natural for an executable world. The interesting question is not whether a model can answer one tax question. It is whether we can build a dial that creates the next thousand tax tasks at the right difficulty.</p><p>That is the new training data business. Not more data for its own sake: better worlds, better tasks, and better feedback.</p><h2>Source Notes</h2><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nzodtGccEM">Don't Just Fine-tune the Agent, Tune the Environment</a></strong> (ICLR 2026 Poster): scarce high-quality LLM-agent training data, SFT overfitting, RL cold start, structured curriculum, environment augmentation, and fine-grained progress rewards.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=Uro84w2xz5">Supervised Reinforcement Learning: From Expert Trajectories to Step-wise Reasoning</a></strong> (ICLR 2026 Poster): RLVR failure when correct solutions are rarely sampled; step-wise rewards and richer signal even when rollouts are incorrect.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ZmGirmNJqE">Search Self-Play: Pushing the Frontier of Agent Capability without Supervision</a></strong> (ICLR 2026 Poster): RLVR dependence on well-crafted tasks and ground truth; generated agentic task difficulty is hard to control.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ZgCCDwcGwn">AgentGym-RL</a></strong> (ICLR 2026 Oral): external environment interaction for long-horizon agent training; staged short-to-long interaction strategy.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=QRSeFZfu8E">Generalizable End-to-End Tool-Use RL with Synthetic CodeGym</a></strong> (ICLR 2026 Poster): static coding problems converted into diverse, verifiable, controllable multi-turn tool-use environments.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nHrYBGujps">BIRD-INTERACT</a></strong> (ICLR 2026 Oral): database plus knowledge base, metadata, user simulator, CRUD tasks, execution feedback, and executable tests.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=cj1MAx7lKs">OpenApps</a></strong> (ICLR 2026 Oral): configurable app worlds, thousands of app variations, and reliability changes across environment variations.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=x54wwB6QvL">Scaling Laws Revisited: Modeling the Role of Data Quality in Language Model Pretraining</a></strong> (ICLR 2026 Poster): data-quality parameter extending Chinchilla-style scaling beyond model size and dataset volume.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=8KcjEygedc">Why Less is More (Sometimes): A Theory of Data Curation</a></strong> (ICLR 2026 Poster): curation by difficulty and correctness; conditions where smaller curated datasets can outperform full datasets.</p></div></div>]]></content:encoded></item><item><title><![CDATA[Reinforcement Learning and Ontologies at ICLR 2026]]></title><description><![CDATA[The previous post argued that the next scarce asset is not generic text or generic generation.]]></description><link>https://nikolaosvasiloglou.substack.com/p/reinforcement-learning-and-ontologies</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/reinforcement-learning-and-ontologies</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Fri, 14 Aug 2026 12:02:36 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!z4fN!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!z4fN!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!z4fN!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!z4fN!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:1223200,&quot;alt&quot;:&quot;A paper-boat navigator follows a winding path through four checked rule cards toward a distant map.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator follows a winding path through four checked rule cards toward a distant map." title="A paper-boat navigator follows a winding path through four checked rule cards toward a distant map." srcset="https://substackcdn.com/image/fetch/$s_!z4fN!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!z4fN!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2002559a-8c5a-4d2c-9e58-fc71ed745b03_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Rules verify each step of a journey instead of checking only the destination.</figcaption></figure></div><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!GyWb!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!GyWb!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!GyWb!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/f480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:156327,&quot;alt&quot;:&quot;A paper-boat navigator sails a red dashed route past four islands: a trophy for outcome reward, a dotted path of checked steps for process trace, connected rule cards for ontology rules, and an anchor with a checkmark for grounded verification.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator sails a red dashed route past four islands: a trophy for outcome reward, a dotted path of checked steps for process trace, connected rule cards for ontology rules, and an anchor with a checkmark for grounded verification." title="A paper-boat navigator sails a red dashed route past four islands: a trophy for outcome reward, a dotted path of checked steps for process trace, connected rule cards for ontology rules, and an anchor with a checkmark for grounded verification." srcset="https://substackcdn.com/image/fetch/$s_!GyWb!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!GyWb!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ff480e64a-3fb5-4ba9-81d6-9c4e30d79aff_3200x1800.png 1456w" sizes="100vw"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Process traces and ontology rules make verification richer than a final-answer reward alone.</figcaption></figure></div><p>The previous post argued that the next scarce asset is not generic text or generic generation. It is the ability to build worlds. This post is about what happens once those worlds exist: agents need to practice in them, fail in them, and be graded by them.</p><h2>The Short Version</h2><p>If you only read one section, read this one.</p><ul><li><p><strong>A gym is a world you can run.</strong> The agent acts, the world changes, and a reward or verifier says whether the action helped. In modern agent systems the gym is a terminal, a browser, a database, a workflow, a simulated lab, or a container full of tools.</p></li><li><p><strong>The grader is not a scoring script &#8212; it is part of the world.</strong> Benchmark design is now partly security work: a good task has to resist agents that hard-code answers, delete failing tests, or satisfy the grader while violating the intent.</p></li><li><p><strong>The answer is not the artifact; the trace is the artifact.</strong> A correct final answer can come from broken reasoning, so a cluster of ICLR 2026 papers moves reward inside the trace &#8212; step-level logical supervision, Lean as a process oracle, step-wise expert rewards, and detectors for reward hacking hidden behind benign-looking chains of thought.</p></li><li><p><strong>The next strategic skill is specifying correctness.</strong> Requirements, constraints, invariants, permissions, completion criteria, failure conditions. In a world of learning agents, whoever can define what counts as correct may have more leverage than whoever writes the code that attempts the task.</p></li><li><p><strong>Enterprises need a different verification language than Lean.</strong> A business is data and rules &#8212; customers, contracts, approvals, exceptions &#8212; not a theorem. Executable ontologies are one plausible candidate: a gym with semantics and a verifier with memory.</p></li></ul><h2>What's in This Post</h2><p>Each section opens with the short version, then a deeper dive with the paper-level evidence. Read the short versions for the argument; stay for the deep dives when you want the receipts.</p><ol><li><p><strong>Gyms, Rollouts, and Graders</strong> &#8212; the vocabulary, and what a modern agent task actually contains.</p></li><li><p><strong>The Grader Is Part of the World</strong> &#8212; why fair grading turned into security work.</p></li><li><p><strong>The Answer Is Not the Artifact</strong> &#8212; the process-supervision cluster at ICLR 2026.</p></li><li><p><strong>Specifying Correctness</strong> &#8212; Lean, and the skill that outlasts writing the code.</p></li><li><p><strong>The Enterprise Turn: Executable Ontologies</strong> &#8212; why business worlds need a different language.</p></li><li><p><strong>The Bottom Line</strong> &#8212; RL needs gyms, gyms need graders, graders need specifications.</p></li></ol><h2>Gyms, Rollouts, and Graders</h2><p><strong>The short version:</strong> A gym is a world you can run: the agent takes an action, the world changes, the agent observes the result, and a reward or verifier says whether the action helped. The sequence of thoughts, tool calls, observations, file edits, queries, errors, and recoveries is the rollout, or trajectory &#8212; the thing the model actually did, not the story it tells afterward. Modern agent tasks package this as a container, an instruction, tests, and a reference solution.</p><h3>Deeper dive: what a Harbor-style task contains</h3><p>In simple games this structure is obvious. In modern agent systems, the gym may be a terminal, a browser, a database, a calendar-style workflow, a simulated lab, or a Docker container full of tools and services.</p><p>Terminal-Bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a>, an ICLR 2026 paper, describes tasks with a containerized environment, an instruction, tests, and a reference solution. The paper uses the Harbor task format and harness. In the Harbor documentation<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a>, the same idea appears as <code>instruction.md</code>, <code>task.toml</code>, an <code>environment/</code> directory, a <code>solution/</code> path, and <code>tests/</code>. The agent receives the task, acts inside the environment, modifies files or services if needed, and leaves behind a final state. The grader then checks whether that final state satisfies the task.</p><p>This design is straightforward in outline until you try to make it fair.</p><h2>The Grader Is Part of the World</h2><p><strong>The short version:</strong> The grader must know whether the agent solved the task or merely passed the tests. A task is not good if the agent can inspect hidden future commits, exploit a fixture, hard-code the answer, delete a failing test, or produce a file that satisfies the grader while violating the real intent. Terminal-Bench devotes real machinery to specificity, solvability, and anti-cheating checks. That is the first ICLR lesson for enterprise AI: a grader is not a scoring script. It is part of the world.</p><h3>Deeper dive: benchmark design as security work</h3><p>Terminal-Bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a> asks whether a task is specific, solvable, and resistant to cheating, and the paper describes integrity and anti-cheating checks as a first-class part of task construction. The reason is straightforward once you take agents seriously as optimizers: any gap between "what the grader measures" and "what the task intends" is an exploitable surface, and a capable agent will find it faster than a human task author will notice it.</p><h2>The Answer Is Not the Artifact</h2><p><strong>The short version:</strong> Even if the final answer is correct, the reasoning may be wrong. A program can pass a test because two bugs cancel. A model can produce the right SQL result after misunderstanding the schema. An agent can repair a service by disabling a feature. A robot can hit a goal state through an unsafe path. In each case the final state is real, but it does not tell us whether the process should be learned, repeated, or trusted. A cluster of ICLR 2026 papers responds by moving reward inside the trace.</p><h3>Deeper dive: the process-supervision cluster</h3><p>LogicReward<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a> says this directly. Existing training methods often depend on outcome-based feedback, and that can produce correct answers with flawed reasoning. LogicReward<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a>'s response is to enforce step-level logical correctness with a theorem prover. It converts natural-language reasoning into formal structure, softens the brittleness of exact formalization with a method it calls Autoformalization with Soft Unification, and then gives the model reward at the level of logical steps rather than only the final answer.</p><p>Reinforcement Learning with Verifiable Rewards Implicitly Incentivizes Correct Reasoning<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a> is the optimistic side of the story. It argues that RLVR can encourage correct reasoning even when rewards are based only on answer correctness, and it introduces CoT-Pass@K to account for both final answers and intermediate reasoning. This matters: RLVR is powerful precisely because it turns verifiable tasks into training signal.</p><p>But other papers show why the final answer is not enough. Process-Verified Reinforcement Learning for Theorem Proving via Lean<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> uses Lean itself as a symbolic process oracle. Lean does not merely say whether the proof eventually worked. Its elaboration can mark locally sound tactics and the earliest failing step, giving dense, verifier-grounded credit signals rooted in type theory. That is a particularly informative grader: it knows the target, knows the formal rules, and can localize failure.</p><p>Supervised Reinforcement Learning<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> makes the credit-assignment problem even clearer. It points out that RLVR can fail when correct solutions are rarely sampled, then proposes smoother step-wise rewards based on expert actions. The important line for business is that richer learning signals can help even when all rollouts are incorrect. A rollout that gets 98 percent of the way there should not be indistinguishable from nonsense. If we cannot give partial credit, we waste training compute and lose diagnostic information.</p><p>TRACE addresses the failure mode. In Is it Thinking or Cheating?<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a>, the authors study implicit reward hacking: the model exploits a loophole in the reward function while its chain-of-thought appears benign. TRACE detects this by truncating reasoning and checking how quickly a model can still obtain reward. If a shortcut gives the reward with surprisingly little reasoning effort, the model may be gaming the world rather than solving the task.</p><p>From Verifiable Dot to Reward Chain<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a> gives the broader framing. RLVR works well in math and code when a final answer can be checked, but single-dot supervision becomes inefficient and reward-hacking-prone when tasks are open-ended. Their proposed move is a reward chain: more ordered, structured feedback than one bit at the end.</p><p>This is the ICLR 2026 verification story in one sentence: the answer is not the artifact; the trace is the artifact.</p><h2>Specifying Correctness</h2><p><strong>The short version:</strong> Lean is the sharpest example of a machine-checkable world because mathematics has unusually clean rules &#8212; a tactic either elaborates or it does not. The broader lesson is not that Lean should replace mainstream programming. It is that knowledge work can move from prose that sounds right to artifacts machines can check, and that the next strategic skill may be the ability to specify correctness: requirements, constraints, invariants, permissions, completion criteria, and failure conditions.</p><h3>Deeper dive: from prose to checkable artifacts</h3><p>A proof assistant can say what follows from what. This is why the recent book The Proof in the Code<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a> is relevant here. Kevin Hartnett traces the birth and rise of Lean as a proof assistant, but the broader relevance is not only mathematical culture. It is the idea that knowledge work can move from prose that sounds right to artifacts that machines can check.</p><p>Python and Java still matter. SQL still matters. But in a world of learning agents, the person who can define what counts as correct may have more leverage than the person who only writes the code that attempts the task.</p><h2>The Enterprise Turn: Executable Ontologies</h2><p><strong>The short version:</strong> Lean is beautiful for mathematical and logical worlds, but an enterprise is not a theorem. It is data: customers, suppliers, products, contracts, inventory, invoices, policies, approvals, events, exceptions, forecasts, and decisions. That world's language has to compute over data and express rules, not just proofs &#8212; closer to SQL, but with semantics, constraints, and executable business logic. An executable ontology fits that description: a gym with semantics and a verifier with memory.</p><h3>Deeper dive: the conference evidence, and a procurement example</h3><p>BIRD-INTERACT<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-10" href="#footnote-10" target="_self">10</a> is the conference evidence that this enterprise-adjacent world is already visible. It takes text-to-SQL out of the static prompt setting and puts it into a dynamic environment with databases, hierarchical knowledge bases, metadata, a user simulator, CRUD tasks, execution feedback, and executable tests. That is much closer to a real business assistant. The agent must ask questions, inspect state, recover from mistakes, and satisfy operational requirements.</p><p>CodeGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a> and AgentGym-RL<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a> show the same pattern from different directions. CodeGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a> converts static coding problems into interactive, verifiable, controllable multi-turn tool-use environments. AgentGym-RL<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a> treats multi-turn interaction across diverse environments as the substrate for training LLM agents. The pattern is not "more prompts." The pattern is executable worlds with verifiers.</p><p>This is where ontologies re-enter the story. An executable ontology is a model of the entities, relationships, rules, constraints, and actions of a domain that a machine can run against. The previous post framed digital twins, ontologies, and gyms as three dialects of the same idea. One way to put the synthesis is this: an executable ontology can function as a gym with semantics and a verifier with memory.</p><p>That view is already being built on: executable ontologies for decision worlds, where the agent must reason over data, rules, relationships, and constraints rather than only generate text. The point is not to replace Lean. It is to recognize that enterprises need a different verification language: one that can express the correctness conditions of a business process, compute over live data, and explain why an action is allowed, useful, or wrong.</p><p>Imagine training an agent to manage procurement. A Harbor-style task can give it a container and a grader. A Lean-style proof can teach us the value of formal verification. But the procurement world itself needs suppliers, contracts, budgets, approval chains, substitutions, delivery risks, tariff rules, and historical outcomes. That is ontology territory. The grader should not merely ask whether the final purchase order exists. It should know whether the agent followed policy, respected budget authority, handled exceptions, chose valid substitutions, and produced an auditable decision trace.</p><h2>The Bottom Line</h2><p>That is the practical synthesis. RL needs gyms. Gyms need graders. Graders need specifications. Specifications need languages. For math, Lean is one answer. For code, tests and type systems are part of the answer. For enterprise AI, executable ontologies are one plausible candidate because the enterprise world is made of data and rules.</p><p>The future of agent training will not be only better models. It will be better worlds that can say, with precision, what happened inside the rollout.</p><h2>Source Notes</h2><p><strong>Harbor-style tasks, rollouts, and grading</strong></p><ul><li><p><strong><a href="https://openreview.net/pdf?id=a7Qa4CcHak">Terminal-Bench PDF</a></strong>, for Harbor task format, task formulation, final-state tests, specificity, solvability, integrity, and anti-cheating audit details</p></li></ul><p><strong>RLVR, process verification, and reward hacking</strong></p><ul><li><p><strong><a href="https://openreview.net/forum?id=sE8DCSJTzd">Exploration vs Exploitation: Rethinking RLVR through Clipping, Entropy, and Spurious Reward</a></strong>, ICLR 2026 Poster</p></li></ul><p><strong>Background sources requested for this topic</strong></p><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=a7Qa4CcHak">Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces</a></strong>, ICLR 2026 Poster/conference paper</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p><strong><a href="https://www.harborframework.com/docs/tasks">Harbor Task Structure documentation</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=IRhYVOKFe0">LogicReward: Incentivizing LLM Reasoning via Step-Wise Logical Supervision</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=jGbRWwIidy">Reinforcement Learning with Verifiable Rewards Implicitly Incentivizes Correct Reasoning in Base LLMs</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=P00k4DFaXF">Process-Verified Reinforcement Learning for Theorem Proving via Lean</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=Uro84w2xz5">Supervised Reinforcement Learning: From Expert Trajectories to Step-wise Reasoning</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=Gk7gLAtVDO">Is it Thinking or Cheating? Detecting Implicit Reward Hacking by Measuring Reasoning Effort</a></strong>, ICLR 2026 Oral</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ZumVIktGbt">From Verifiable Dot to Reward Chain: Harnessing Verifiable Reference-based Rewards for Reinforcement Learning of Open-ended Generation</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p><strong><a href="https://www.quantabooks.org/books/the-proof-in-the-code/">The Proof in the Code by Kevin Hartnett, Quanta Books</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-10" href="#footnote-anchor-10" class="footnote-number" contenteditable="false" target="_self">10</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nHrYBGujps">BIRD-INTERACT: Re-imagining Text-to-SQL Evaluation via Lens of Dynamic Interactions</a></strong>, ICLR 2026 Oral</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-11" href="#footnote-anchor-11" class="footnote-number" contenteditable="false" target="_self">11</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=QRSeFZfu8E">Generalizable End-to-End Tool-Use RL with Synthetic CodeGym</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-12" href="#footnote-anchor-12" class="footnote-number" contenteditable="false" target="_self">12</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ZgCCDwcGwn">AgentGym-RL: An Open-Source Framework to Train LLM Agents for Long-Horizon Decision Making via Multi-Turn RL</a></strong>, ICLR 2026 Oral</p></div></div>]]></content:encoded></item><item><title><![CDATA[The New Worlds of AI and How to Build Them]]></title><description><![CDATA[The previous post ended with a question: if interaction data is the scarce resource, how do we create the worlds that produce it?]]></description><link>https://nikolaosvasiloglou.substack.com/p/the-new-worlds-of-ai-and-how-to-build</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/the-new-worlds-of-ai-and-how-to-build</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Thu, 13 Aug 2026 22:25:19 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!TTzL!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!TTzL!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!TTzL!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!TTzL!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:1207910,&quot;alt&quot;:&quot;A paper-boat navigator assembles connected knowledge nodes, curated layers, mirrored structures, and a small practice arena.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator assembles connected knowledge nodes, curated layers, mirrored structures, and a small practice arena." title="A paper-boat navigator assembles connected knowledge nodes, curated layers, mirrored structures, and a small practice arena." srcset="https://substackcdn.com/image/fetch/$s_!TTzL!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!TTzL!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F39e2b1f9-f1d1-4d1c-aa1e-ae0592ba1965_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Knowledge, curation, twins, and practice environments become one buildable world.</figcaption></figure></div><p>The previous post ended with a question: if interaction data is the scarce resource, how do we create the worlds that produce it?</p><p>This is where ICLR 2026 becomes much more practical than the phrase "AI research conference" suggests. The strategic question is not whether models can generate more candidates. They can. It is whether those candidates can survive a world. Can the molecule be synthesized? Can the antibody bind? Can the robot manipulate the object? Can the database assistant recover after an execution error? Can the calendar agent respect a policy, a user's preference, and a conflicting meeting constraint?</p><p>The model does not need only a bigger imagination. It needs a world that can push back.</p><h2>The Short Version</h2><p>If you only read one section, read this one.</p><ul><li><p><strong>World-building, not scraping, is the new bottleneck.</strong> Generating candidates is cheap; narrowing them to a set that can be manufactured, tested, measured, and fed back is not. The field is building simulated laboratory worlds before claiming autonomous discovery &#8212; and agents given real scientific software still reach only 15% success.</p></li><li><p><strong>Physical labs don't generalize.</strong> You can rent a biology lab. You cannot cheaply create a new company, a port, a supply chain shock, a regulatory crisis, or zero gravity on demand. The real problem is broader: making domains executable.</p></li><li><p><strong>Three ways to do it.</strong> (1) Encode knowledge into the world so models don't rediscover physics from scratch. (2) Curate the data rather than piling up tokens &#8212; quality is now a variable in the scaling equation, not background. (3) Build executable models of the domain: digital twins, ontologies, and gyms.</p></li><li><p><strong>Those three words describe one thing: a world you can run.</strong> Enterprises say digital twin, knowledge-representation people say ontology, RL people say gym. The "game" is now a calendar, a SQL database, a browser, a lab bench, a robot workcell, or a procurement process.</p></li><li><p><strong>This reframes the moat.</strong> Generic generation keeps getting cheaper. The durable capability is knowing your domain well enough to build the world in which intelligence can practice.</p></li></ul><h2>What's in This Post</h2><p>Each numbered direction opens with the short version, then a deeper dive with the paper-level evidence. Read the short versions for the argument; stay for the deep dives when you want the receipts.</p><ol><li><p><strong>Why Physical Labs Are Not Enough</strong> &#8212; where automated science stops generalizing.</p></li><li><p><strong>Encode Knowledge Into the World</strong> &#8212; symmetry, boundary conditions, and world models that know things.</p></li><li><p><strong>Curate the Data, Not Just the Dataset</strong> &#8212; quality as a first-class scaling variable.</p></li><li><p><strong>Build Digital Twins, Ontologies, and Gyms</strong> &#8212; the environment turn, and what it means for enterprises.</p></li></ol><h2>Why Physical Labs Are Not Enough</h2><p><strong>The short version:</strong> AI4Science makes the bottleneck visible. The automated lab &#8212; model proposes, robot executes, instruments measure, measurement becomes new data &#8212; is real enough to shape the research agenda, but ICLR's clearest signal is more careful: the field is building simulated laboratory worlds first. And physical labs only cover part of the need. We cannot cheaply instantiate a company, a port, a supply chain shock, or a regulatory crisis whenever we need training data. The world-building problem is bigger than robotics.</p><h3>Deeper dive: simulated labs and the 15% ceiling</h3><p>AutoBio<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a> is a simulation and benchmark for robotic automation in digital biology laboratories. It digitizes lab instruments, adds specialized physics plugins, renders dynamic instrument interfaces and transparent materials, and evaluates biologically grounded manipulation tasks. The headline is not "robots can now do all biology." It is that scientific automation needs a structured, high-precision world before the model's actions mean anything.</p><p>ScienceBoard<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a> makes the same point from the software side of science. It gives agents realistic scientific workflows with professional software across domains such as biochemistry, astronomy, and geoinformatics. The benchmark has 169 human-curated and validated tasks, and current agents reach only 15 percent overall success. That number is useful because it keeps demos in perspective. An agent that can describe science is not yet an agent that can do science inside a real workflow.</p><p>Even when a world is physically possible, collecting enough safe, diverse, repeatable interaction data may be too slow or too expensive. There are three main ways to make domains executable instead.</p><h2>1. Encode Knowledge Into the World</h2><p><strong>The short version:</strong> Reduce the data you need by putting more of the domain into the model, simulator, or environment itself. Physics is the cleanest case because physics has laws: if a system must conserve mass, obey a boundary condition, or respect a symmetry, the model should not rediscover that from scratch. ICLR evidence is pointed &#8212; equivariant models that leverage task symmetry scale better, and for neural operators, small amounts of boundary-diverse data beat large amounts of same-family data. If the missing variable defines the world, volume inside the wrong world does not fix the problem.</p><h3>Deeper dive: symmetry, boundaries, and world models</h3><p>In Scaling Laws and Symmetry<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a>, the authors study neural force fields for interatomic potentials and show that architectures with more symmetry expressivity have different scaling exponents. Equivariant models that leverage task symmetry scale better than non-equivariant models. The operational lesson is important: as systems scale, we should not simply leave fundamental inductive biases such as symmetry for the model to discover.</p><p>Boundary conditions tell the same story in a different dialect. Neural Operators Fail to Generalize Across Boundary Condition Families<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a> shows that a neural operator trained on one boundary-condition family can look good in distribution and then fail badly when the boundary family changes. In the paper's experiments, adding much more same-family training data helps less than adding small amounts of boundary-diverse data.</p><p>Robotics gives the embodied version of this lesson. WorldGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> treats a world model as an environment for policy evaluation because real robot testing is costly and handcrafted simulators are hard to improve. Ctrl-World<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> makes the same economic point: rigorous robot evaluation requires many real rollouts, and systematic improvement requires corrective expert data. Their proposed alternative is a controllable world model in which policies can roll out in imagination space, with synthetic successful trajectories used to improve the policy.</p><p>This is not a claim that simulation has solved robotics. WorldGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> explicitly notes that highly realistic object interaction remains hard. The point is more useful: when real interaction is expensive, the world has to carry domain knowledge. It must encode geometry, contact, action, time, failure, and reward in a way that produces useful pressure on the policy.</p><p>That is the first direction: build worlds that know things.</p><h2>2. Curate the Data, Not Just the Dataset</h2><p><strong>The short version:</strong> Stop treating "data" as a scalar. Chinchilla-style scaling laws captured a real relationship, but they treated quality as background. It isn't anymore &#8212; recent work adds an explicit data-quality parameter to the scaling equation, and higher-quality data can reduce model size and compute requirements. For enterprises this matters because your useful data is not the web. It is policy, exceptions, procedures, approvals, contracts, and outcomes: compressed world knowledge. But quality is not universal, so curation has to be tied to the world being built.</p><h3>Deeper dive: quality as a scaling variable</h3><p>Scaling Laws Revisited<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a> makes this explicit. The paper extends the Chinchilla framework with a dimensionless data-quality parameter, <code>Q</code>, so loss can be modeled as a function of model size, data volume, and quality. In its experiments, higher-quality data can reduce model size and compute requirements. That does not mean "small data beats big data" as a slogan. It means the unit is not just the token. The unit is useful signal.</p><p>Why Less is More (Sometimes)<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a> attacks the same issue from theory. It asks when it is better to use less data and gives conditions under which curated subsets can outperform full datasets. The important business lesson is not that curation always wins. It is that the answer depends on data quality, task difficulty, noise, and the selection strategy. Once those variables matter, the company that knows which examples belong in the world has an advantage over the company that only owns a larger pile.</p><p>This matters for enterprise AI because most useful enterprise data is not "the web." It is policy, exceptions, procedures, approvals, customer histories, product hierarchies, supplier constraints, contract terms, taxonomies, SLAs, decisions, and outcomes. Much of it is boring in exactly the way valuable data is boring: it encodes how the organization actually works. A manual, a textbook, a protocol, or a policy document is not merely text. It is compressed world knowledge.</p><p>The catch is that quality is not universal. A clean textbook may be high-quality data for a tutor and weak data for an agent that must operate a warehouse. A database schema may be high-quality data for an analytics agent and nearly useless for a procurement negotiation unless it is connected to suppliers, terms, approval rules, and exception handling.</p><p>That is the second direction: build worlds out of the right data, not just more data.</p><h2>3. Build Digital Twins, Ontologies, and Gyms</h2><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!EANC!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!EANC!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!EANC!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!EANC!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!EANC!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!EANC!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/d4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:154303,&quot;alt&quot;:&quot;A paper-boat navigator sails a red dashed route past four islands: a node graph for knowledge models, sorting bins for data curation, twin mirrored houses for digital twins, and a training ring with a dumbbell and flag for agent gyms.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator sails a red dashed route past four islands: a node graph for knowledge models, sorting bins for data curation, twin mirrored houses for digital twins, and a training ring with a dumbbell and flag for agent gyms." title="A paper-boat navigator sails a red dashed route past four islands: a node graph for knowledge models, sorting bins for data curation, twin mirrored houses for digital twins, and a training ring with a dumbbell and flag for agent gyms." srcset="https://substackcdn.com/image/fetch/$s_!EANC!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!EANC!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!EANC!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!EANC!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fd4ecdc86-e3f3-48ed-90a1-ff6e9a8bf5f5_3200x1800.png 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Explicit knowledge and curated data support digital twins and gyms where agents can learn and be tested.</figcaption></figure></div><p><strong>The short version:</strong> This is the direction enterprises should watch most closely: build executable models of the domain. Digital twin, ontology, and gym come from different traditions but point at the same thing &#8212; a world you can run. The RL community has had this intuition for years; what is new is the pattern moving into software, science, databases, and enterprise workflows. ICLR 2026 shows the environment turn clearly, including a result worth internalizing: agent reliability can vary by more than 50 percent across variations of the same app. A useful world is not one pretty clone. It is a generator of realistic variations that expose brittle behavior.</p><h3>Deeper dive: the environment turn</h3><p>Enterprises like "digital twin" because it sounds like a model of a product, factory, supply chain, port, grid, or company. Database and knowledge-representation people like "ontology" because it means entities, relationships, constraints, rules, and semantics. Reinforcement-learning people like "gym" because it means an environment where an agent can act, observe, receive reward, fail, reset, and try again. These are not identical traditions, but strategically they point at the same thing.</p><p>A game environment makes the intuition obvious. An agent plays Pong, takes actions, observes pixels or state, receives reward, and learns. The "game" is now a calendar, a messenger, a SQL database, a browser, a lab bench, a robot workcell, or a procurement process.</p><p>Several ICLR 2026 papers show this environment turn. AgentGym-RL<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a> is an open-source framework for training LLM agents across diverse, realistic multi-turn environments, and the paper argues that agents need expanded external interaction rather than internal reasoning alone. CodeGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-10" href="#footnote-10" target="_self">10</a> converts static coding problems into interactive, verifiable, controllable tool-use environments. Environment Tuning<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a> goes further and says: do not only fine-tune the agent; tune the environment, using curriculum, environment augmentation, and progress rewards so agents can learn from problem instances without pre-collected expert trajectories.</p><p>BIRD-INTERACT<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a> is a strong enterprise-adjacent example. It turns text-to-SQL from a single prompt into a dynamic world: database, hierarchical knowledge base, metadata, user simulator, clarification, execution errors, CRUD tasks, and executable tests. This looks much closer to production work than a static benchmark because real data assistants have to ask questions, recover from mistakes, and satisfy changing requirements.</p><p>OpenApps<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-13" href="#footnote-13" target="_self">13</a> shows how cheap these worlds can become when the domain is modeled correctly. It provides configurable messenger, calendar, maps, and other app environments that run on a single CPU and can generate thousands of versions. The paper finds that agent reliability can vary by more than 50 percent across app variations.</p><p>WALT<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-14" href="#footnote-14" target="_self">14</a> offers another clue. Instead of forcing a web agent to reason through every click and layout change, it reverse-engineers latent website functionality into deterministic callable tools such as search, filter, create, edit, and delete. That is a world abstraction. It takes messy surface interaction and exposes stable operations.</p><h3>Deeper dive: what this means for your company</h3><p>This is why the word ontology matters for enterprise readers. A good ontology is not just a diagram. It is a portable, structured model of what exists, how things relate, what actions are possible, what constraints must hold, and what counts as success. If it is executable, it can become a gym. If it mirrors a company or process, it can become a digital twin.</p><p>The enterprise version should be concrete. As an illustrative pattern, you do not need to train an agent directly on a live Google Calendar or Slack workspace to teach scheduling behavior. You can model users, calendars, rooms, policies, conflicts, permissions, messages, deadlines, and success conditions. You can reset the world, create adversarial cases, simulate ambiguity, and verify outcomes. You do not need to touch the real procurement system to train negotiation or approval behavior. You can model suppliers, catalogs, contracts, budget constraints, approval chains, exceptions, and audit rules.</p><p>That is not a toy version of the company. If the ontology captures the right entities, relationships, constraints, and verifiers, it is the version of the company that an agent can safely learn against.</p><h2>The Bottom Line</h2><p>In our view, this reframes the moat. The moat is less about generic generation, which large labs are likely to keep making cheaper, and more about the ability to build useful worlds: scientific worlds with physics and measurement, robotic worlds with contact and action, enterprise worlds with rules and consequences, and software worlds with tools and tests.</p><p>In the old data regime, the question was "how much can we scrape?" In the new regime, the question is "what worlds can we make executable?"</p><p>That question is harder, but it is also more strategic. It turns AI from a model-buying exercise into an organizational capability: knowing the domain well enough to build the world in which intelligence can practice.</p><h2>Source Notes</h2><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=UUE6HEtjhu">AutoBio: A Simulation and Benchmark for Robotic Automation in Digital Biology Laboratory</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=bJvwJahJeF">ScienceBoard: Evaluating Multimodal Autonomous Agents in Realistic Scientific Workflows</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=qyjaVda7t2">Scaling Laws and Symmetry, Evidence from Neural Force Fields</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p><strong><a href="https://openreview.net/pdf?id=zp8ILnE0Or">Neural Operators Fail to Generalize Across Boundary Condition Families</a></strong>, ICLR 2026 conference paper PDF</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=hidBHy1CAw">WorldGym: World Model as An Environment for Policy Evaluation</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=748bHL2BAv">Ctrl-World: A Controllable Generative World Model for Robot Manipulation</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=x54wwB6QvL">Scaling Laws Revisited: Modeling the Role of Data Quality in Language Model Pretraining</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=8KcjEygedc">Why Less is More (Sometimes): A Theory of Data Curation</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ZgCCDwcGwn">AgentGym-RL: An Open-Source Framework to Train LLM Agents for Long-Horizon Decision Making via Multi-Turn RL</a></strong>, ICLR 2026 Oral</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-10" href="#footnote-anchor-10" class="footnote-number" contenteditable="false" target="_self">10</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=QRSeFZfu8E">Generalizable End-to-End Tool-Use RL with Synthetic CodeGym</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-11" href="#footnote-anchor-11" class="footnote-number" contenteditable="false" target="_self">11</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nzodtGccEM">Don't Just Fine-tune the Agent, Tune the Environment</a></strong>, ICLR 2026 Poster</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-12" href="#footnote-anchor-12" class="footnote-number" contenteditable="false" target="_self">12</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nHrYBGujps">BIRD-INTERACT: Re-imagining Text-to-SQL Evaluation via Lens of Dynamic Interactions</a></strong>, ICLR 2026 Oral</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-13" href="#footnote-anchor-13" class="footnote-number" contenteditable="false" target="_self">13</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=cj1MAx7lKs">OpenApps: Simulating Environment Variations to Measure UI Agent Reliability</a></strong>, ICLR 2026 Oral</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-14" href="#footnote-anchor-14" class="footnote-number" contenteditable="false" target="_self">14</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=cgIDqcJcoI">WALT: Web Agents that Learn Tools</a></strong>, ICLR 2026 Poster</p></div></div>]]></content:encoded></item><item><title><![CDATA[The End of Cheap Data: Rethinking the Scaling Laws]]></title><description><![CDATA[The first era of modern AI scaling was built on cheap passive data.]]></description><link>https://nikolaosvasiloglou.substack.com/p/the-end-of-cheap-data-rethinking</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/the-end-of-cheap-data-rethinking</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Thu, 13 Aug 2026 22:19:19 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!H_Tt!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!H_Tt!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!H_Tt!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!H_Tt!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/a4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:1051934,&quot;alt&quot;:&quot;A sparse cartoon shows many dots passing through three filters while a paper-boat navigator inspects a few colored shapes.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A sparse cartoon shows many dots passing through three filters while a paper-boat navigator inspects a few colored shapes." title="A sparse cartoon shows many dots passing through three filters while a paper-boat navigator inspects a few colored shapes." srcset="https://substackcdn.com/image/fetch/$s_!H_Tt!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!H_Tt!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Fa4bf62dc-423d-4dfb-b8e7-d1fce02c77b7_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">A large field of data is filtered into a few valuable verified training items.</figcaption></figure></div><p>The first era of modern AI scaling was built on cheap passive data. We turned the web, code repositories, math solutions, and worked explanations into training data, and the model learned to predict, complete, translate, summarize, answer, and eventually reason in sequences. That era produced an enormous leap.</p><p>But the next bottleneck is not another trillion scraped tokens. The next bottleneck is interaction data: data from models acting in external worlds, receiving feedback, recovering from mistakes, and learning which actions actually change the state of a system.</p><p>This is why the scaling-laws conversation has to change. The old question was how loss improves as model size, compute, and token count increase. That question still matters. But the strategic question is now different: how do we create enough high-quality worlds for models to act in?</p><h2>The Short Version</h2><p>If you only read one section, read this one.</p><ul><li><p><strong>Cheap passive data built the first scaling era; interaction data is the next bottleneck.</strong> The valuable corpus is no longer text about the world &#8212; it is traces of models acting inside worlds that push back.</p></li><li><p><strong>Software makes the gap concrete.</strong> Writing code is text generation; development is an action loop. Frontier agents score below 65% on Terminal-Bench's hard terminal tasks, and a model at 74.4% on SWE-bench Verified completes only 11.0% of FeatureBench's feature-level tasks.</p></li><li><p><strong>The gap repeats in every world.</strong> Web data science: 15% agent success where humans reach 90%. Messy tool use: no model above 15% across 57 LLMs. Interactive database work: GPT-5 under 17%. Robotics and science are harder still &#8212; six drug-design models generated roughly 120,000 molecules and exactly one showed even very weak activity.</p></li><li><p><strong>Synthetic data only helps when verified.</strong> Model-collapse results show that retraining on unverified self-generated data can degrade models; the moat is not "generate synthetic data" but verified world generation.</p></li><li><p><strong>The new scaling question is conditional:</strong> more useful data, from richer worlds, with better verifiers, stronger provenance, and tighter feedback loops &#8212; not just more parameters, compute, and tokens.</p></li></ul><h2>What's in This Post</h2><p>Each section opens with the short version, then a deeper dive with the paper-level evidence. Read the short versions for the argument; stay for the deep dives when you want the receipts.</p><ol><li><p><strong>Code Is Not Development</strong> &#8212; the cleanest demonstration that passive data ran out.</p></li><li><p><strong>The Biggest Data Gap</strong> &#8212; digital worlds: web tasks, tools, and databases.</p></li><li><p><strong>Physical Worlds Are Harder</strong> &#8212; robotics as a data-world construction problem.</p></li><li><p><strong>Scientific Worlds Are Stranger</strong> &#8212; why generated candidates are not results.</p></li><li><p><strong>Is the Future Dull?</strong> &#8212; the new data economy, and the synthetic-data caveat.</p></li><li><p><strong>The New Scaling Question</strong> &#8212; what replaces "more tokens."</p></li></ol><h2>Code Is Not Development</h2><p><strong>The short version:</strong> Writing code is text generation. Development is an action loop: terminal, files, dependencies, tests, errors, logs, retries. New benchmarks finally measure the difference, and the numbers are stark &#8212; frontier models and agents score below 65% on Terminal-Bench's hard terminal tasks, and a model reported at 74.4% on SWE-bench Verified succeeds on only 11.0% of FeatureBench's end-to-end feature tasks. The easy corpus was code as text. The valuable corpus is now code as interaction.</p><h3>Deeper dive: the anatomy of an action loop</h3><p>That distinction sounds obvious to engineers, but it has not always been obvious in AI benchmarks. A model that writes a plausible patch has not necessarily done development. A development agent has to survive contact with the environment.</p><p>Terminal-Bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a> was a quiet but consequential shift because it made this distinction concrete. At the simplest level, a Terminal-Bench-style setup has four pieces:</p><ol><li><p>An LLM that proposes actions.</p></li><li><p>A terminal agent or harness that executes those actions and returns observations.</p></li><li><p>A sandboxed environment, typically container-like, where the work can run safely.</p></li><li><p>A verifier, usually tests, that decides whether the task succeeded.</p></li></ol><p>That is a different kind of data. It is not just "here is a problem and here is the answer." It is an action loop. The model proposes. The world responds. The model adapts. The verifier judges.</p><p>Terminal-Bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a> makes this explicit: 89 hard tasks in command-line environments, each with a unique environment, a human-written solution, and tests. The reported result is not "frontier models are solved." It is the opposite: frontier models and agents score below 65%.</p><p>FeatureBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a> sharpens the point. It evaluates end-to-end feature development across 200 tasks and 3,825 executable environments from 24 repositories. The striking number is the gap: a model reported at 74.4% on SWE-bench Verified succeeds on only 11.0% of FeatureBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a> tasks. That is the difference between generating code and doing development.</p><h2>The Biggest Data Gap</h2><p><strong>The short version:</strong> The major next data gap in AI is not another archive of web pages. It is traces of generative models interacting with worlds. Across web data science, messy real-world tool use, and interactive database work, agents that look strong on static benchmarks collapse in interactive ones. That pattern matters for enterprises because businesses are not prompt-answer machines &#8212; they are worlds, with rules, exceptions, roles, approvals, state changes, and external systems. Operating inside them requires world data.</p><h3>Deeper dive: three digital worlds, one pattern</h3><p>WebDS<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a> tests agents on 870 web-based data science tasks across 29 websites. These tasks require finding data, navigating heterogeneous sources, cleaning, analyzing, and summarizing. One reported browser agent completes 80% of tasks on WebVoyager but only 15% on WebDS<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a>, while humans reach about 90%.</p><p>WildToolBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a> shows the same story for tools. It is built around messy user behavior: compositional tasks, implicit intent spread across turns, and instruction transitions where the user changes direction or clarifies. Across 57 LLMs, no model exceeds 15% accuracy.</p><p>BIRD-INTERACT<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> brings the point into databases and enterprise-adjacent workflows. Static text-to-SQL is not enough when real database work requires clarification, execution feedback, CRUD operations, metadata, knowledge bases, user simulation, and executable tests. The full benchmark contains 600 tasks with up to 11,796 dynamic interactions. GPT-5 completes 8.67% in the conversational setting and 17.00% in the more open-ended agentic setting.</p><p>Those benchmarked settings are narrower than a full enterprise, but the pattern matters for enterprises because supply chains, sales operations, marketing workflows, procurement, finance, compliance, support, and database operations are governed by rules, exceptions, roles, approvals, state changes, and external systems. A model that can write a paragraph about procurement is not yet an agent that can operate inside procurement.</p><h2>Physical Worlds Are Harder</h2><p><strong>The short version:</strong> The terminal is comparatively cheap to reset. The browser adds state and ambiguity. Robotics makes the feedback loop materially more expensive: geometry, friction, contact, latency, perception, hardware failures. ICLR 2026 shows robotics becoming a data-world construction problem &#8212; learned world models standing in as environments, video foundation models becoming policies, and imagined trajectories becoming training data.</p><h3>Deeper dive: world models as environments</h3><p>WorldGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> treats a learned world model as an environment for robot policy evaluation. The motivation is blunt: real-world testing is costly, and handcrafted simulators require manual effort to improve realism. WorldGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> uses an action-conditioned video generation model as a proxy environment for robot policies, then compares policy success in that generated world with real-world success.</p><p>Cosmos Policy<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a> pushes from the other direction. It adapts a video foundation model into a robot policy that can generate actions, future states, and values. The point is not just to make robot videos look good. The model uses future-state and value predictions for planning action trajectories.</p><p>Ctrl-World<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a> makes the synthetic-data loop even more explicit. It trains a controllable multi-view world model on the DROID dataset, with 95,000 trajectories across 564 scenes, and uses imagined successful trajectories for supervised fine-tuning. The paper reports a 44.7% policy-success improvement.</p><p>That is the pattern: world model, rollout, action, observation, verifier, correction. Robotics makes it visible because the real world is stubborn.</p><h2>Scientific Worlds Are Stranger</h2><p><strong>The short version:</strong> A generative model can propose molecules, proteins, materials, hypotheses, equations, and experimental plans. But a generated candidate is not a result &#8212; it has to survive the world it claims to describe. The cleanest warning from ICLR 2026: six public structure-based drug-design models generated roughly 120,000 molecules, and exactly one purchased compound showed even very weak activity. Agents given real scientific software reach only 15% success. These are not failures of sentence fluency; they are failures of world binding. AI for science is a world-building problem: simulators, verifiers, experimental interfaces, expert-curated tasks, and representations that respect the domain.</p><h3>Deeper dive: chemistry, biology, physics, and the verification bottleneck</h3><p>In chemistry, the GEM workshop talk on experimental assessments of structure-based drug design<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a> is the cleanest warning: six public structure-based drug-design models generated roughly 120,000 molecules, about 99% were removed by pre-filtering, 47 compounds were bought, and only one showed very weak inhibition or binding at 100 micromolar. The useful artifact is not the generated ligand alone; it is the ligand after screening, purchase, assay, and measurement. The same point appears in paper form through SYNC<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-10" href="#footnote-10" target="_self">10</a>, which treats synthesizability as part of structure-based drug design, and 3DCS<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a>, which stresses that molecular representations must be sensitive to 3D conformation rather than only valid as graphs or strings.</p><p>In biology, the lesson is constraint. Constrained Diffusion for Protein Design<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a> builds hard structural constraints into protein generation, because rejection after unconstrained sampling is expensive and often wasteful. Genomics adds a different version of the same warning: ContrastiveBiVI<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-13" href="#footnote-13" target="_self">13</a> focuses on perturbation effects in transcriptional dynamics, while Taking The Easy Way Out<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-14" href="#footnote-14" target="_self">14</a> shows how single-cell foundation models can learn local correlation shortcuts instead of biological mechanism when the masking objective ignores gene co-regulation.</p><p>Physics and PDEs make the failure mode even sharper. One Operator to Rule Them All?<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-15" href="#footnote-15" target="_self">15</a> argues that neural PDE solvers can behave like boundary-indexed families rather than a single boundary-agnostic operator. Residual-Spectrum Diagnostics<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-16" href="#footnote-16" target="_self">16</a> makes a related evaluation point: low solution error does not necessarily mean the prediction satisfies the governing equation. And text-trained LLMs extrapolating PDE dynamics<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-17" href="#footnote-17" target="_self">17</a> show that even when models mimic spatiotemporal patterns, multi-step rollouts accumulate error over the horizon.</p><p>ScienceBoard<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-18" href="#footnote-18" target="_self">18</a> is a useful bridge because it evaluates multimodal autonomous agents in realistic scientific workflows. It gives agents professional scientific software and 169 human-curated tasks across areas such as biochemistry, astronomy, and geoinformatics. Despite strong backbones, evaluated agents reach only 15% overall success. The same verification problem appears in Silent Squad<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-19" href="#footnote-19" target="_self">19</a>, where multimodal agents use real scientific software, and in DSGym<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-20" href="#footnote-20" target="_self">20</a>, which reports that many existing data-science benchmark tasks can be shortcut without using the actual data files.</p><p>Scientific hypothesis generation has the same asymmetry at a larger scale. The Verification Bottleneck<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-21" href="#footnote-21" target="_self">21</a> frames the problem directly: AI can generate plausible claims faster than scientific institutions can verify them. Work on controlled synthetic causal worlds<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-22" href="#footnote-22" target="_self">22</a> adds a narrower but important point: plausible causal stories are not enough when hidden confounding and mechanism composition matter.</p><p>Physics benchmarks show another side of the gap. CMT-Benchmark<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-23" href="#footnote-23" target="_self">23</a> tests expert-level condensed matter theory, including quantum many-body and classical statistical mechanics. Across 17 models, average performance is reported as 11.4% plus or minus 2.1%, with GPT-5 solving 30%. CMPhysBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-24" href="#footnote-24" target="_self">24</a> contains more than 520 graduate-level condensed matter physics questions and reports the best model at 29% accuracy.</p><p>The model can produce plausible symbols, but the scientific world has constraints: conservation laws, symmetries, boundary conditions, non-commuting operators, experimental noise, physical feasibility, and domain-specific notions of correctness.</p><h2>Is the Future Dull?</h2><p><strong>The short version:</strong> No. The end of cheap data is not the end of progress; it is the beginning of a more interesting data economy. Building interactive worlds used to require tedious expert labor. Modern coding assistants can multiply expert throughput &#8212; drafting tasks, scaffolding tests, building simulators and harnesses &#8212; as long as experts stay in the loop. But the caveat is real: unverified synthetic data can collapse models. The moat is not "generate synthetic data." The moat is verified world generation.</p><h3>Deeper dive: the economics, and the collapse caveat</h3><p>In the old regime, creating interactive data for a specialized world required tedious expert labor. Someone had to define the scenarios, build the environment, write the tests, produce solutions, validate edge cases, and coordinate between domain experts and engineers. That was slow, expensive, and hard to scale.</p><p>Modern coding assistants may change the economics. They can draw on broad technical corpora to draft tasks, generate candidate solutions, scaffold tests, create simulators, write harnesses, and help experts iterate. They are not reliable enough to replace experts. But they are useful enough to multiply expert throughput when experts remain in the loop.</p><p>This matters because the most valuable synthetic data is not arbitrary text. It is verified interaction data. It is a scenario plus a world plus a solution trace plus an evaluator. It is a task where the model must act and the environment can push back.</p><p>There is a caveat. Synthetic data can hurt. ICLR 2026 has multiple model-collapse papers warning that retraining on self-generated data can degrade models unless the data is mixed, verified, or anchored by real signal. "Escaping Model Collapse via Synthetic Data Verification<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-25" href="#footnote-25" target="_self">25</a>" shows that an external verifier, human or stronger model, can prevent collapse in the studied setting, but imperfect verification can plateau or reverse gains. Work on optimal mixing ratios<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-26" href="#footnote-26" target="_self">26</a> similarly argues that real data remains necessary even when synthetic data helps.</p><h2>The New Scaling Question</h2><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!laBq!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!laBq!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!laBq!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!laBq!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!laBq!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!laBq!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/2d84513a-4876-4892-93db-115d0077184d_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:152590,&quot;alt&quot;:&quot;A paper-boat navigator sails a red dashed route past four islands: a heap of raw data for volume scaling, a funnel for data quality, rising steps for task difficulty, and a magnifier with a checkmark for verification.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator sails a red dashed route past four islands: a heap of raw data for volume scaling, a funnel for data quality, rising steps for task difficulty, and a magnifier with a checkmark for verification." title="A paper-boat navigator sails a red dashed route past four islands: a heap of raw data for volume scaling, a funnel for data quality, rising steps for task difficulty, and a magnifier with a checkmark for verification." srcset="https://substackcdn.com/image/fetch/$s_!laBq!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!laBq!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!laBq!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!laBq!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F2d84513a-4876-4892-93db-115d0077184d_3200x1800.png 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">The next scaling frontier combines data quality, calibrated difficulty, and verification.</figcaption></figure></div><p>The old scaling law was built around more: more parameters, more compute, more tokens.</p><p>The emerging scaling question is conditional: more useful data, from richer worlds, with better verifiers, stronger provenance, and tighter feedback loops.</p><p>That is why data-quality scaling papers matter. "Scaling Laws Revisited<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-27" href="#footnote-27" target="_self">27</a>" explicitly adds data quality to the scaling equation. "Why Less is More (Sometimes)<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-28" href="#footnote-28" target="_self">28</a>" gives conditions under which curated subsets can beat full datasets. Common Corpus<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-29" href="#footnote-29" target="_self">29</a> shows that even pretraining data is becoming a provenance and licensing problem, not merely a scraping problem.</p><p>The strategic implication is clarifying. Generic generation is likely to keep becoming cheaper, but durable value sits upstream and downstream of generation: tools that create worlds, environments that return feedback, verifiers that distinguish success from plausible but incorrect outputs, and curated traces that bind models to specific realities.</p><p>For software, that world is the terminal, the repository, the test suite, the UI, and the deployment environment. For robotics, it is the physical scene and the world model that can safely approximate it. For science, it is the simulator, lab, equation, assay, microscope, telescope, or quantum/chemical system. For the enterprise, it is the process graph: roles, rules, approvals, schemas, exceptions, contracts, and operational state.</p><p>We are no longer only in the business of harvesting data. We are in the business of making worlds legible to models.</p><p>The next question is the hard one: how do we create these worlds?</p><h2>Source Notes</h2><p>Reasoning, data curation, and scaling:</p><ul><li><p><strong><a href="https://openreview.net/forum?id=7xjoTuaNmN">OpenThoughts, "Data Recipes for Reasoning Models"</a></strong></p></li><li><p><strong><a href="https://openreview.net/forum?id=FdkPOHlChS">"Softmax Transformers are Turing-Complete"</a></strong></p></li></ul><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=a7Qa4CcHak">Terminal-Bench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=41xrZ3uGuI">FeatureBench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=7cHhcrbr6x">WebDS</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=yz7fL5vfpn">WildToolBench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=nHrYBGujps">BIRD-INTERACT</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=hidBHy1CAw">WorldGym</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=wPEIStHxYH">Cosmos Policy</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=748bHL2BAv">Ctrl-World</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39063706">Joshua Heston, "Experimental Assessments of Structure-Based Drug Design," ICLR 2026 GEM workshop, SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-10" href="#footnote-anchor-10" class="footnote-number" contenteditable="false" target="_self">10</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=y1tPw4Uuzg">SYNC, "Measuring and Advancing Synthesizability in Structure-Based Drug Design"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-11" href="#footnote-anchor-11" class="footnote-number" contenteditable="false" target="_self">11</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=JAb0y8lkqL">3DCS, "Datasets and Benchmark for Evaluating Conformational Sensitivity in Molecular Representations"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-12" href="#footnote-anchor-12" class="footnote-number" contenteditable="false" target="_self">12</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=kkvqVRu2Zy">"Constrained Diffusion for Protein Design with Hard Structural Constraints"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-13" href="#footnote-anchor-13" class="footnote-number" contenteditable="false" target="_self">13</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=yLfHPt1LZa">ContrastiveBiVI, "Exploring Perturbation Effects on Transcriptional Dynamics"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-14" href="#footnote-anchor-14" class="footnote-number" contenteditable="false" target="_self">14</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=KlDSNIvt9A">"Taking The Easy Way Out: When Single-Cell Foundation Models Learn Shortcuts Instead of Biology"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-15" href="#footnote-anchor-15" class="footnote-number" contenteditable="false" target="_self">15</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=lDjWQ9UxRy">"One Operator to Rule Them All? On Boundary-Indexed Operator Families in Neural PDE Solvers"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-16" href="#footnote-anchor-16" class="footnote-number" contenteditable="false" target="_self">16</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=e49Aef9YtL">"What Does a Neural PDE Solver Really Learn? A Residual-Spectrum Diagnostic"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-17" href="#footnote-anchor-17" class="footnote-number" contenteditable="false" target="_self">17</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=wdFdObhQnG">"Text-Trained LLMs Can Zero-Shot Extrapolate PDE Dynamics"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-18" href="#footnote-anchor-18" class="footnote-number" contenteditable="false" target="_self">18</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=bJvwJahJeF">ScienceBoard</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-19" href="#footnote-anchor-19" class="footnote-number" contenteditable="false" target="_self">19</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39061693">"Silent Squad: Evaluating Multi-Modal Autonomous Agents in Realistic Scientific Workflows," ICLR 2026 SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-20" href="#footnote-anchor-20" class="footnote-number" contenteditable="false" target="_self">20</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=Iimpdn9umQ">DSGym, "A Standardized and Holistic Framework for Advancing Data Science Agents"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-21" href="#footnote-anchor-21" class="footnote-number" contenteditable="false" target="_self">21</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=XDukYDg7Qj">"The Verification Bottleneck: Managing Trust in Post-AGI Science"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-22" href="#footnote-anchor-22" class="footnote-number" contenteditable="false" target="_self">22</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=2sn6LSR9CB">"Revisiting Causal Reasoning in Language Models through Controlled Synthetic Worlds"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-23" href="#footnote-anchor-23" class="footnote-number" contenteditable="false" target="_self">23</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=pX6B28ynNh">CMT-Benchmark</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-24" href="#footnote-anchor-24" class="footnote-number" contenteditable="false" target="_self">24</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=3d0FRYx0D0">CMPhysBench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-25" href="#footnote-anchor-25" class="footnote-number" contenteditable="false" target="_self">25</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=yfk6c39omW">"Escaping Model Collapse via Synthetic Data Verification"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-26" href="#footnote-anchor-26" class="footnote-number" contenteditable="false" target="_self">26</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=9RdhTvYbX0">"Preventing Model Collapse Under Overparametrization"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-27" href="#footnote-anchor-27" class="footnote-number" contenteditable="false" target="_self">27</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=x54wwB6QvL">"Scaling Laws Revisited: Modeling the Role of Data Quality in Language Model Pretraining"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-28" href="#footnote-anchor-28" class="footnote-number" contenteditable="false" target="_self">28</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=8KcjEygedc">"Why Less is More (Sometimes): A Theory of Data Curation"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-29" href="#footnote-anchor-29" class="footnote-number" contenteditable="false" target="_self">29</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=0wSlFpMsGb">Common Corpus, "The Largest Collection of Ethical Data for LLM Pre-Training"</a></strong></p></div></div>]]></content:encoded></item><item><title><![CDATA[The Academic Trilogy of AI, ICLR Part 1: Why Should I Care?]]></title><description><![CDATA[Every year, AI now has a public rhythm.]]></description><link>https://nikolaosvasiloglou.substack.com/p/the-academic-trilogy-of-ai-iclr-part</link><guid isPermaLink="false">https://nikolaosvasiloglou.substack.com/p/the-academic-trilogy-of-ai-iclr-part</guid><dc:creator><![CDATA[Nikolaos Vasiloglou]]></dc:creator><pubDate>Thu, 13 Aug 2026 12:22:03 GMT</pubDate><enclosure url="https://substackcdn.com/image/fetch/$s_!xf00!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png" length="0" type="image/jpeg"/><content:encoded><![CDATA[<div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!xf00!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!xf00!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!xf00!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!xf00!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!xf00!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!xf00!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png" width="1672" height="941" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:941,&quot;width&quot;:1672,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:975246,&quot;alt&quot;:&quot;A hand-drawn paper-boat navigator studies research signals and follows a red dotted path toward a map at sunrise.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:false,&quot;topImage&quot;:true,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A hand-drawn paper-boat navigator studies research signals and follows a red dotted path toward a map at sunrise." title="A hand-drawn paper-boat navigator studies research signals and follows a red dotted path toward a map at sunrise." srcset="https://substackcdn.com/image/fetch/$s_!xf00!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 424w, https://substackcdn.com/image/fetch/$s_!xf00!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 848w, https://substackcdn.com/image/fetch/$s_!xf00!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 1272w, https://substackcdn.com/image/fetch/$s_!xf00!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F30d08b66-3625-4382-999c-bd478fae77e2_1672x941.png 1456w" sizes="100vw" fetchpriority="high"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">A paper navigator turns scattered conference signals into a route toward a roadmap.</figcaption></figure></div><p>Every year, AI now has a public rhythm. ICLR opens the year in the spring. ICML carries the summer. NeurIPS closes the year in December.</p><p>Fifteen years ago, most executives, investors, and product leaders could safely ignore this calendar. The conferences were academic meetings, the papers needed a PhD colleague to translate, and the market delay could be years. That bargain is gone. In 2026, the work presented at these venues arrives already wrapped in code, benchmarks, datasets, evaluation harnesses, agent environments, leaderboards, safety tests, and workflow abstractions. The lag between "research result" and "thing a developer can run, copy, benchmark against, or get pressured by" is now measured in months, sometimes weeks.</p><div class="subscription-widget-wrap-editor" data-attrs="{&quot;url&quot;:&quot;https://nikolaosvasiloglou.substack.com/subscribe?&quot;,&quot;text&quot;:&quot;Subscribe&quot;,&quot;language&quot;:&quot;en&quot;}" data-component-name="SubscribeWidgetToDOM"><div class="subscription-widget show-subscribe"><div class="preamble"><p class="cta-caption">Thanks for reading! Subscribe for free to receive new posts and support my work.</p></div><form class="subscription-widget-subscribe"><input type="email" class="email-input" name="email" placeholder="Type your email&#8230;" tabindex="-1"><input type="submit" class="button primary" value="Subscribe"><div class="fake-input-wrapper"><div class="fake-input"></div><div class="fake-button"></div></div></form></div></div><p>This post makes the case for why that matters to you, using two trends that visibly changed shape in the six months between NeurIPS 2025 and ICLR 2026.</p><h2>The Short Version</h2><p>If you only read one section, read this one.</p><ul><li><p><strong>The trilogy is an early-warning layer.</strong> ICLR, ICML, and NeurIPS are where the field shows its working memory: what researchers think is hard, which benchmarks are losing credibility, which abstractions are becoming standard, and which product categories will soon feel obvious.</p></li><li><p><strong>Scale killed paper-by-paper reading.</strong> ICLR 2026 accepted 5,355 papers, slightly more than the NeurIPS 2025 main track. The useful question is no longer "which paper is important?" but "which weak signals are becoming clusters?"</p></li><li><p><strong>Trend 1: the LLM monolith is breaking into layers.</strong> Attention variants, KV-cache policy, memory, routing, controllers, and adaptive compute are becoming separable systems layers. At NeurIPS 2025 you had to assemble this thesis from scattered fragments; by ICLR 2026 it was a dense, named research program.</p></li><li><p><strong>Trend 2: evaluation is moving from static benchmarks to real tasks.</strong> Executable environments, trajectories, tools, logs, and human baselines are replacing one-prompt-one-score tests. The new benchmarks are humbling: a frontier model that scores 74.4% on SWE-bench Verified succeeds on only 11% of FeatureBench tasks.</p></li><li><p><strong>The answer is to read conferences as data.</strong> Extract papers, claims, benchmarks, and systems into knowledge graphs, compare signals across venues, and watch clusters form before they reach vendor roadmaps.</p></li></ul><h2>What's in This Post</h2><p>Each section below opens with the short version, then offers a deeper dive with the paper-level evidence. Read the short versions for the argument; stay for the deep dives when you want the receipts.</p><ol><li><p><strong>The Scale Has Changed</strong> &#8212; the trilogy by the numbers, and why ICLR now rivals NeurIPS.</p></li><li><p><strong>From Reading Papers to Reading Clusters</strong> &#8212; why the old way of following research broke.</p></li><li><p><strong>Trend 1: The Architectural Monolith Breaks Into Systems Layers</strong> &#8212; the model is becoming one layer in a stack.</p></li><li><p><strong>Trend 2: Evaluation Moves From Static Benchmarks to Real Tasks</strong> &#8212; benchmarks are becoming executable environments, and agents are failing them.</p></li><li><p><strong>Our Answer: Analyze the Conferences as Data</strong> &#8212; how we turn conference corpora into roadmap input.</p></li><li><p><strong>The Bottom Line</strong> &#8212; what to do about it.</p></li></ol><h2>The Scale Has Changed</h2><p><strong>The short version:</strong> ICLR, the youngest of the three conferences, now accepts slightly more main-conference papers than NeurIPS. These are no longer small specialist venues. They are enormous technical markets of ideas, tools, benchmarks, and talent &#8212; and their scale is exactly why they need a new way of being read.</p><h3>Deeper dive: the numbers</h3><p>The history is worth being precise about. NeurIPS was founded in 1987. ICML's named-conference lineage starts in 1993<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-1" href="#footnote-1" target="_self">1</a>, building on earlier International Workshops on Machine Learning. ICLR is much younger; the first ICLR was held in 2013<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-2" href="#footnote-2" target="_self">2</a>.</p><p>So the "trilogy" order is calendar order, not age order: ICLR, then ICML, then NeurIPS. But the calendar order also used to carry a rough intuition about scale. ICLR was the younger spring conference. NeurIPS was the giant year-end finale. That intuition is now outdated.</p><p>NeurIPS 2025 reported<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-3" href="#footnote-3" target="_self">3</a> 21,575 valid main-track submissions and 5,290 accepted papers, a 24.52% acceptance rate, supported by 20,518 reviewers, 1,663 area chairs, and 199 senior area chairs. ICLR 2026 reported<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-4" href="#footnote-4" target="_self">4</a> 19,525 valid submissions, 76,139 reviews, 18,054 reviewers, and 5,355 accepted papers, a 27.4% acceptance rate. By accepted main-conference papers, ICLR 2026 slightly exceeded the NeurIPS 2025 main track.</p><p>That does not mean ICLR is larger in every sense. Attendance data still shows NeurIPS as the larger gathering in 2025: the AI Index / Our World in Data series<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-5" href="#footnote-5" target="_self">5</a> reports 26,382 attendees for NeurIPS 2025, compared with 11,039 for ICLR 2025 and 8,000 for ICML 2025, with the usual caveat that conference attendance can mix in-person, online, and hybrid participation depending on the year.</p><h2>From Reading Papers to Reading Clusters</h2><p><strong>The short version:</strong> The old way to read conference research was paper by paper. That worked when the output was small and the market delay was long. It breaks when one conference accepts more than five thousand papers and the delay compresses from years to months. The skill that matters now is spotting which weak signals are becoming clusters &#8212; and the two trends below show how fast that can happen.</p><p>Between NeurIPS 2025 and ICLR 2026, two examples make the change visible. They are not the only ones; similar six-month shifts appear across data curation, scientific discovery, safety, verification, and small-model deployment. But these two are especially clear because they moved from scattered evidence to dense, explicit, named research programs. A useful way to read them: the first is about model/runtime decomposition, the second about evaluation realism.</p><h2>Trend 1: The Architectural Monolith Breaks Into Systems Layers</h2><p><strong>The short version:</strong> The transformer is being decomposed into separable layers &#8212; attention mechanisms, state-space routing, KV-cache management, memory, model routers, dynamic depth, and test-time compute policies. The model is no longer the whole product; it is becoming one layer in a stack that also includes cache policy, memory, routing, inference budgets, tool interfaces, verifiers, and runtime policy.</p><p>The six-month compression is the story. At NeurIPS 2025 the stack thesis existed, but you had to assemble it yourself from a paper here, a workshop slide there. By ICLR 2026 the pieces were dense enough that the stack surfaced on its own. If your company builds on or buys LLM infrastructure, this is where differentiation is moving.</p><h3>Deeper dive: the paper trail from NeurIPS to ICLR</h3><p>At NeurIPS 2025, the LLM-stack thesis was already present, but it was not yet sitting in one obvious place. Routing Mamba<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-6" href="#footnote-6" target="_self">6</a> treated sequence modeling as a mixture of state-space dynamics and sparse expert routing. Gated Attention<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-7" href="#footnote-7" target="_self">7</a> made the attention block itself more controllable by adding gates to softmax attention. ChunkKV<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-8" href="#footnote-8" target="_self">8</a>, KVzip<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-9" href="#footnote-9" target="_self">9</a>, and RetrievalAttention<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-10" href="#footnote-10" target="_self">10</a> made KV state a memory-hierarchy problem: compress it, evict it, retrieve it, or move indexes off the critical GPU path.</p><p>Twilight<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-11" href="#footnote-11" target="_self">11</a> treated attention budget as something to allocate adaptively. REFRAG<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-12" href="#footnote-12" target="_self">12</a> exploited attention sparsity to accelerate RAG decoding, while AdmTree<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-13" href="#footnote-13" target="_self">13</a> compressed long context through adaptive semantic trees. A-Mem<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-14" href="#footnote-14" target="_self">14</a> made agent memory a dynamic, evolving subsystem and compared against MemGPT alongside other memory baselines. KVCOMM<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-15" href="#footnote-15" target="_self">15</a> pushed the KV cache into multi-agent communication, where shared context becomes a systems object rather than a hidden implementation detail.</p><p>The workshop evidence points in the same direction. In the Lock-LLM workshop<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-16" href="#footnote-16" target="_self">16</a>, one slide summarized the scaling shift as "multi-dimensional": architecture was moving from dense to sparse, capabilities toward long context and multimodality, and inference toward test-time scaling.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-17" href="#footnote-17" target="_self">17</a></p><p>Those papers and workshops did not all announce the same thesis. That is the point. A human analyst had to connect state-space routing, gated attention, KV eviction and retrieval, semantic context compression, agent memory, attention sparsity, context reuse, and test-time compute into one pattern: the model was becoming one layer in a stack.</p><p>By ICLR 2026, the signal had become much denser. Mamba-3<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-18" href="#footnote-18" target="_self">18</a> was an oral paper built around an "inference-first" state-space design. From Collapse to Control<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-19" href="#footnote-19" target="_self">19</a> analyzed long-context failure in hybrid Mamba-Transformer models and proposed a bridge between Transformer position scaling and Mamba state dynamics. Scaling Laws Meet Model Architecture<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-20" href="#footnote-20" target="_self">20</a> moved scaling-law discussion beyond parameters and data into architecture choices such as hidden size, the MLP-to-attention ratio, and grouped-query attention.</p><p>The attention and KV layer was no longer a side topic. Draft-based Approximate Inference<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-21" href="#footnote-21" target="_self">21</a> used small draft models to guide KV dropping and prompt compression. FusedKV<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-22" href="#footnote-22" target="_self">22</a>, FreeKV<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-23" href="#footnote-23" target="_self">23</a>, KVTC<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-24" href="#footnote-24" target="_self">24</a>, and ICaRus<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-25" href="#footnote-25" target="_self">25</a> treated KV state as something to reconstruct, retrieve, compress, store, share, or reuse across models. IceCache<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-26" href="#footnote-26" target="_self">26</a> combined semantic token clustering with PagedAttention, while FASA<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-27" href="#footnote-27" target="_self">27</a> and ThinKV<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-28" href="#footnote-28" target="_self">28</a> made sparse attention and cache compression adaptive to frequency, queries, or reasoning state. vAttention<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-29" href="#footnote-29" target="_self">29</a> and Race Attention<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-30" href="#footnote-30" target="_self">30</a> pushed sparse and linear attention toward deployable regimes with verification or kernel-level evidence.</p><p>The routing and control layer was visible too. Universal Model Routing<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-31" href="#footnote-31" target="_self">31</a> selected among models dynamically. LD-MoLE<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-32" href="#footnote-32" target="_self">32</a> made LoRA expert allocation token-dependent and layer-wise. DND<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-33" href="#footnote-33" target="_self">33</a> selectively reprocessed critical tokens through dynamic depth. Strategic Scaling of Test-Time Compute<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-34" href="#footnote-34" target="_self">34</a> treated inference as an allocation problem, giving more compute to harder queries and less to easier ones. vCache<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-35" href="#footnote-35" target="_self">35</a> added verified prompt caching, and the MemAgents workshop made the vocabulary explicit with sessions on controllable parametric memory<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-36" href="#footnote-36" target="_self">36</a> and data-centric memory for LLM agents<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-37" href="#footnote-37" target="_self">37</a>.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-38" href="#footnote-38" target="_self">38</a></p><p>Read together, this is no longer a story about one bigger model or one undifferentiated transformer call. It is a story about specialized layers: architecture, cache, memory, routing, inference budgets, tool interfaces, verifiers, and runtime policy.</p><h2>Trend 2: Evaluation Moves From Static Benchmarks to Real Tasks</h2><p><strong>The short version:</strong> The old evaluation pattern was one prompt, one answer, one score. That is increasingly inadequate &#8212; for agents especially, but also for coding, web data science, tool use, and long-horizon planning. The new benchmarks look like real work: executable environments, changing state, trajectories, tools, logs, human baselines, and partial failures.</p><p>The new benchmarks are also humbling. Frontier models stay below 65% on Terminal-Bench's hard terminal tasks. A model at 74.4% on SWE-bench Verified succeeds on only 11.0% of FeatureBench's feature-level tasks. A browser agent that scores 80% on WebVoyager falls to 15% on WebDS, where humans reach almost 90%. If you are deciding whether to trust a benchmark claim or move an agent workflow into production, this gap between old and new benchmarks is material input to the decision.</p><h3>Deeper dive: from scattered signs to a named transition</h3><p>At NeurIPS 2025, the evidence for this benchmark-realism turn was present, but distributed. OpenCUA<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-39" href="#footnote-39" target="_self">39</a> treated computer-use agents as a scalable open-source training and evaluation problem. TheAgentCompany<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-40" href="#footnote-40" target="_self">40</a> framed workplace simulation as an agent benchmark. AgentIF<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-41" href="#footnote-41" target="_self">41</a> tested instruction following in tool-constrained agentic scenarios.</p><p>The spoken conference material pointed in the same direction. The NeurIPS social on evaluating agentic systems<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-42" href="#footnote-42" target="_self">42</a> argued that static and single-turn metrics fail for autonomous planning, tool use, and long-horizon behavior. In AI4NextG, the agent was described as a task-level system with persistent goals, memory, real-time feedback, tool and action selection, and continuous replanning.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-43" href="#footnote-43" target="_self">43</a> The same source captured the practical QA problem in blunt terms: once agentic AI is "not deterministic," the verification process has to change.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-44" href="#footnote-44" target="_self">44</a></p><p>Again, this was not absent at NeurIPS. It was scattered. The papers, social session, and workshop evidence were pointing beyond "can the model answer?" toward "can the system act?" But the benchmark infrastructure was still forming.</p><p>By ICLR 2026, the trend was much harder to miss. The conference program itself was naming the transition. The Agents in the Wild workshop<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-45" href="#footnote-45" target="_self">45</a> scheduled invited talks titled From LLMs to Agents: The Evaluation Challenge<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-46" href="#footnote-46" target="_self">46</a> and Compound AI Systems: Design Patterns for Secure Multi-Agent Deployments<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-47" href="#footnote-47" target="_self">47</a>. The DATA-FM workshop carried The Art and Science of Benchmarking Agents<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-48" href="#footnote-48" target="_self">48</a>. Our local ICLR transcript pass summarized the same shift: evaluation was moving from static, single-turn benchmarks toward multi-step, multi-agent systems that use tools and engineering environments, while the difficulty of evaluating agents was growing alongside the ease of building them.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-49" href="#footnote-49" target="_self">49</a></p><p>The broader ICLR source pack made clear why static benchmarks are failing. Its cross-topic synthesis describes a shift from "model as product" to "AI as composed, governed system," and the executive summary argues that progress and risk now sit at the system level, not only at the model level.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-50" href="#footnote-50" target="_self">50</a> The reasoning track made a parallel evaluation point: chain-of-thought and longer inference were being reframed as controlled systems involving data, RL/RLVR, verifiers, scaffolds, memory, retrieval, tools, and cost-aware test-time compute.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-51" href="#footnote-51" target="_self">51</a> Agentic Context Engineering<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-52" href="#footnote-52" target="_self">52</a> and the MemAgents workshop are useful here because they show why realistic evaluation has to include changing context and memory, not just final answers to static prompts.<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-53" href="#footnote-53" target="_self">53</a></p><p>The paper evidence became more concrete too. Terminal-Bench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-54" href="#footnote-54" target="_self">54</a> evaluated agents on hard terminal tasks in isolated environments, each with a human-written solution and tests, and still found frontier models below 65%. FeatureBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-55" href="#footnote-55" target="_self">55</a> moved software evaluation from patch-level work toward feature-level implementation across 200 tasks and 3,825 runtime environments, reporting that Claude 4.5 Opus, at 74.4% on SWE-bench Verified, succeeded on only 11.0% of FeatureBench tasks. WebDS<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-56" href="#footnote-56" target="_self">56</a> made the same realism point for web data science: Browser Use fell from 80% on WebVoyager to 15% on WebDS, while humans reached almost 90%.</p><p>Computer Agent Arena<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-57" href="#footnote-57" target="_self">57</a> broadened the scope to computer-use agents in dynamic, human-centric settings. TRACE<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-58" href="#footnote-58" target="_self">58</a> pushed toward benchmarks that evolve through recorded and validated trajectories. Holistic Agent Leaderboard<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-59" href="#footnote-59" target="_self">59</a> standardized agent evaluation across hundreds of VMs and inspected logs to find failure modes that raw success rates miss. MCP-SafetyBench<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-60" href="#footnote-60" target="_self">60</a> showed why this is also a safety and governance problem: once agents use real tool servers, evaluation has to cover actions, permissions, and attack surfaces, not just text outputs.</p><p>In December, the field had strong but discontinuous signs that simple static benchmarks were becoming inadequate: agents needed memory, tools, persistent goals, replanning, non-deterministic QA, and more realistic tasks. By April, the signs had become papers, workshops, talks, leaderboards, harnesses, and failure taxonomies. The conference was not merely saying "agents are coming." It was telling us where agents break, which benchmarks are too easy, what new harnesses are being built, and which product claims will need stronger evidence.</p><p>This is exactly why conference analysis has become business analysis. If your company is deciding whether to deploy coding agents, use local models, trust benchmark claims, expose tools through an MCP-like interface, or move agent workflows into production, these papers are not abstract. They are material input to roadmap discussions.</p><h2>Our Answer: Analyze the Conferences as Data</h2><div class="captioned-image-container"><figure><a class="image-link image2 is-viewable-img" target="_blank" href="https://substackcdn.com/image/fetch/$s_!K24P!,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png" data-component-name="Image2ToDOM"><div class="image2-inset"><picture><source type="image/webp" srcset="https://substackcdn.com/image/fetch/$s_!K24P!,w_424,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!K24P!,w_848,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!K24P!,w_1272,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!K24P!,w_1456,c_limit,f_webp,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 1456w" sizes="100vw"><img src="https://substackcdn.com/image/fetch/$s_!K24P!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png" width="3200" height="1800" data-attrs="{&quot;src&quot;:&quot;https://substack-post-media.s3.amazonaws.com/public/images/fae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png&quot;,&quot;srcNoWatermark&quot;:null,&quot;fullscreen&quot;:null,&quot;imageSize&quot;:null,&quot;height&quot;:1800,&quot;width&quot;:3200,&quot;resizeWidth&quot;:null,&quot;bytes&quot;:220045,&quot;alt&quot;:&quot;A paper-boat navigator with a graduation cap gathers dashed signal lines from three small islands and turns them into one red dashed route that leads onto a folded roadmap marked with an X and a flag, under a rising sun.&quot;,&quot;title&quot;:null,&quot;type&quot;:null,&quot;href&quot;:null,&quot;belowTheFold&quot;:true,&quot;topImage&quot;:false,&quot;internalRedirect&quot;:null,&quot;isProcessing&quot;:false,&quot;align&quot;:null,&quot;offset&quot;:false}" class="sizing-normal" alt="A paper-boat navigator with a graduation cap gathers dashed signal lines from three small islands and turns them into one red dashed route that leads onto a folded roadmap marked with an X and a flag, under a rising sun." title="A paper-boat navigator with a graduation cap gathers dashed signal lines from three small islands and turns them into one red dashed route that leads onto a folded roadmap marked with an X and a flag, under a rising sun." srcset="https://substackcdn.com/image/fetch/$s_!K24P!,w_424,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 424w, https://substackcdn.com/image/fetch/$s_!K24P!,w_848,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 848w, https://substackcdn.com/image/fetch/$s_!K24P!,w_1272,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 1272w, https://substackcdn.com/image/fetch/$s_!K24P!,w_1456,c_limit,f_auto,q_auto:good,fl_progressive:steep/https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2Ffae119e2-e4eb-4ddd-a205-c6fd878efeff_3200x1800.png 1456w" sizes="100vw" loading="lazy"></picture><div class="image-link-expand"><div class="pencraft pc-display-flex pc-gap-8 pc-reset"><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container restack-image"><svg aria-hidden="true" width="20" height="20" viewBox="0 0 20 20" fill="none" stroke-width="1.5" stroke="var(--color-fg-primary)" stroke-linecap="round" stroke-linejoin="round" xmlns="http://www.w3.org/2000/svg"><g><path d="M2.53001 7.81595C3.49179 4.73911 6.43281 2.5 9.91173 2.5C13.1684 2.5 15.9537 4.46214 17.0852 7.23684L17.6179 8.67647M17.6179 8.67647L18.5002 4.26471M17.6179 8.67647L13.6473 6.91176M17.4995 12.1841C16.5378 15.2609 13.5967 17.5 10.1178 17.5C6.86118 17.5 4.07589 15.5379 2.94432 12.7632L2.41165 11.3235M2.41165 11.3235L1.5293 15.7353M2.41165 11.3235L6.38224 13.0882"></path></g></svg></button><button tabindex="0" type="button" class="pencraft pc-reset pencraft icon-container view-image"><svg xmlns="http://www.w3.org/2000/svg" width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-maximize2 lucide-maximize-2"><polyline points="15 3 21 3 21 9"></polyline><polyline points="9 21 3 21 3 15"></polyline><line x1="21" x2="14" y1="3" y2="10"></line><line x1="3" x2="10" y1="21" y2="14"></line></svg></button></div></div></div></a><figcaption class="image-caption">Conference signals become useful only after evidence synthesis connects them to roadmap choices.</figcaption></figure></div><p>The answer is not for every executive to read five thousand papers. That is not a viable operating model.</p><p>We have been building a more systematic way to read these conferences. We treat the conference as a corpus. We extract papers, claims, benchmarks, methods, datasets, systems, and evaluation results. We organize them into knowledge graphs. We compare signals across conferences. We ask which ideas are isolated, which ones are becoming clusters, and which ones are crossing from research vocabulary into product vocabulary.</p><p>That is how the two trends above become useful before they are obvious. At NeurIPS 2025, cache policy, architecture, routing, computer-use agents, workplace simulation, and agentic evaluation looked like related but separate fragments. By ICLR 2026, they had hardened into two clearer signals: the model/runtime is decomposing into layers, and evaluation is moving from static tests to realistic complex tasks with environments, tools, verifiers, logs, memory, process supervision, and safety boundaries. The field had moved from "can the model answer?" to "can the system act reliably?"</p><p>That is the reason to care about ICLR. Not because every paper matters. Most individual papers will not matter to your business. But the clusters matter. The direction of benchmark design matters. The failure modes matter. The abstractions that repeat across hundreds of papers matter.</p><p>If you wait until those abstractions are packaged into vendor roadmaps, you have given up some early signal.</p><h2>The Bottom Line</h2><p>You cannot be in the AI business and ignore ICLR, ICML, and NeurIPS.</p><p>You do not need to treat the conferences as infallible. You do not need to accept every benchmark, chase every model, or pretend that academic incentives perfectly match product reality. But you do need a disciplined way to watch them.</p><p>The frontier now moves through papers, code, benchmarks, data, and systems at the same time. The companies that learn to read that motion early will make better product bets, ask better vendor questions, hire against the right capabilities, and avoid being surprised when "research" turns into "everyone expects this feature" six months later.</p><p>ICLR is no longer just a conference for researchers. It is a map of what the AI market is about to argue over next.</p><h2>Source Notes</h2><p>Conference scale and history:</p><ul><li><p><strong><a href="https://neurips.cc/Conferences/2026">NeurIPS 2026 conference page</a></strong></p></li></ul><ul><li><p><strong><a href="https://icml.cc/Conferences/2025/CallForPapers">ICML 2025 call for papers</a></strong></p></li><li><p><strong><a href="https://csconfstats.xoveexu.com/conferences/icml/2025/">CS Conf Stats, ICML 2025</a></strong></p></li></ul><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-1" href="#footnote-anchor-1" class="footnote-number" contenteditable="false" target="_self">1</a><div class="footnote-content"><p><strong><a href="https://icml.cc/2016/index.html%3Fp%3D175.html">ICML 2016 history page</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-2" href="#footnote-anchor-2" class="footnote-number" contenteditable="false" target="_self">2</a><div class="footnote-content"><p><strong><a href="https://iclr.cc/archive/2013/">ICLR 2013 archive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-3" href="#footnote-anchor-3" class="footnote-number" contenteditable="false" target="_self">3</a><div class="footnote-content"><p><strong><a href="https://blog.neurips.cc/2025/09/30/reflections-on-the-2025-review-process-from-the-program-committee-chairs/">NeurIPS, "Reflections on the 2025 Review Process from the Program Committee Chairs"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-4" href="#footnote-anchor-4" class="footnote-number" contenteditable="false" target="_self">4</a><div class="footnote-content"><p><strong><a href="https://blog.iclr.cc/2026/03/31/a-retrospective-on-the-iclr-2026-review-process/">ICLR, "A Retrospective on the ICLR 2026 Review Process"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-5" href="#footnote-anchor-5" class="footnote-number" contenteditable="false" target="_self">5</a><div class="footnote-content"><p><strong><a href="https://ourworldindata.org/grapher/attendance-major-artificial-intelligence-conferences">Our World in Data / AI Index, "Attendance at major AI conferences"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-6" href="#footnote-anchor-6" class="footnote-number" contenteditable="false" target="_self">6</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=lqywifxoo1">Routing Mamba, "Scaling State Space Models with Mixture-of-Experts Projection"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-7" href="#footnote-anchor-7" class="footnote-number" contenteditable="false" target="_self">7</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=1b7whO4SfY">Gated Attention for Large Language Models, "Non-linearity, Sparsity, and Attention-Sink-Free"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-8" href="#footnote-anchor-8" class="footnote-number" contenteditable="false" target="_self">8</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=20JDhbJqn3">ChunkKV, "Semantic-Preserving KV Cache Compression for Efficient Long-Context LLM Inference"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-9" href="#footnote-anchor-9" class="footnote-number" contenteditable="false" target="_self">9</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=JFygzwx8SJ">KVzip, "Query-Agnostic KV Cache Compression with Context Reconstruction"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-10" href="#footnote-anchor-10" class="footnote-number" contenteditable="false" target="_self">10</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=8z3cOVER4z">RetrievalAttention, "Accelerating Long-Context LLM Inference via Vector Retrieval"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-11" href="#footnote-anchor-11" class="footnote-number" contenteditable="false" target="_self">11</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=Ve693NkzcU">Twilight, "Adaptive Attention Sparsity with Hierarchical Top-p Pruning"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-12" href="#footnote-anchor-12" class="footnote-number" contenteditable="false" target="_self">12</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=T16WNvb9lj">REFRAG, "Rethinking RAG based Decoding"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-13" href="#footnote-anchor-13" class="footnote-number" contenteditable="false" target="_self">13</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=l3Qq5MU5VX">AdmTree, "Compressing Lengthy Context with Adaptive Semantic Trees"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-14" href="#footnote-anchor-14" class="footnote-number" contenteditable="false" target="_self">14</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=FiM0M8gcct">A-Mem, "Agentic Memory for LLM Agents"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-15" href="#footnote-anchor-15" class="footnote-number" contenteditable="false" target="_self">15</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=yGOytgjurF">KVCOMM, "Online Cross-context KV-cache Communication for Efficient LLM-based Multi-agent Systems"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-16" href="#footnote-anchor-16" class="footnote-number" contenteditable="false" target="_self">16</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39053628/lockllm-workshop-prevent-unauthorized-knowledge-use-from-large-language-models-deep-dive-into-undistillate-unfinetunable-uncompressible-uneditable-and-unusable">NeurIPS 2025 SlidesLive, "Lock-LLM Workshop: Prevent Unauthorized Knowledge Use from Large Language Models"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-17" href="#footnote-anchor-17" class="footnote-number" contenteditable="false" target="_self">17</a><div class="footnote-content"><p>Conference session and paper evidence from the NeurIPS 2025 analysis pipeline, summarized in <a href="https://drive.google.com/file/d/1tvLtyWGQlfXcEJrusnukJwHnHKzPSZ-1/view">08_language_models.md</a>, <a href="https://drive.google.com/file/d/1Y3N89WzpQP2rP55b1xtEw1yJ-woJPz1P/view">17_llm_scaling_laws_efficiency.md</a>, and <a href="https://drive.google.com/file/d/1__snWpVsnIHh_Ja9lAHJVViR2JWOxcfj/view">02_genai_architectures.md</a> (Google Drive): Lock-LLM slide 248 at 04:30:29 frames scaling across architecture, capabilities, and inference; the same source pass flags attention gates, KV compression/retrieval, sparse attention, and long-context memory mechanisms across the NeurIPS program. Public anchors include the Lock-LLM SlidesLive recording<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-61" href="#footnote-61" target="_self">61</a>, Lock-LLM workshop site<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-62" href="#footnote-62" target="_self">62</a>, and the NeurIPS OpenReview papers cited in Trend 1.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-18" href="#footnote-anchor-18" class="footnote-number" contenteditable="false" target="_self">18</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=HwCvaJOiCj">Mamba-3, "Improved Sequence Modeling using State Space Principles"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-19" href="#footnote-anchor-19" class="footnote-number" contenteditable="false" target="_self">19</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=MjmORKLHUI">"From Collapse to Control: Understanding and Extending Context Length in Emerging Hybrid Models via Universal Position Interpolation"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-20" href="#footnote-anchor-20" class="footnote-number" contenteditable="false" target="_self">20</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=0TmVqOpBbK">"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-21" href="#footnote-anchor-21" class="footnote-number" contenteditable="false" target="_self">21</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=0vbYakkECY">Draft-based Approximate Inference for LLMs</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-22" href="#footnote-anchor-22" class="footnote-number" contenteditable="false" target="_self">22</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=4pivvEJiCl">Reconstructing KV Caches with Cross-Layer Fusion for Enhanced Transformers</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-23" href="#footnote-anchor-23" class="footnote-number" contenteditable="false" target="_self">23</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=wXAn7orB1H">FreeKV, "Boosting KV Cache Retrieval for Efficient LLM Inference"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-24" href="#footnote-anchor-24" class="footnote-number" contenteditable="false" target="_self">24</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=aNVKROYpLB">KVTC, "KV Cache Transform Coding for Compact Storage in LLM Inference"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-25" href="#footnote-anchor-25" class="footnote-number" contenteditable="false" target="_self">25</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=qrMo6R7lOS">ICaRus, "Identical Cache Reuse for Efficient Multi-Model Inference"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-26" href="#footnote-anchor-26" class="footnote-number" contenteditable="false" target="_self">26</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=yHxSKM9kdr">IceCache, "Memory-Efficient KV-cache Management for Long-Sequence LLMs"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-27" href="#footnote-anchor-27" class="footnote-number" contenteditable="false" target="_self">27</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=FnSgecCEwg">FASA, "Frequency-Aware Sparse Attention"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-28" href="#footnote-anchor-28" class="footnote-number" contenteditable="false" target="_self">28</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=M3CeHnZKNC">ThinKV, "Thought-Adaptive KV Cache Compression for Efficient Reasoning Models"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-29" href="#footnote-anchor-29" class="footnote-number" contenteditable="false" target="_self">29</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=zzTDulLys0">vAttention, "Verified Sparse Attention via Sampling"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-30" href="#footnote-anchor-30" class="footnote-number" contenteditable="false" target="_self">30</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=RR8Lh8RHgA">RACE Attention, "A Strictly Linear-Time Attention Layer for Training Long Sequences"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-31" href="#footnote-anchor-31" class="footnote-number" contenteditable="false" target="_self">31</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ka82fvJ5f1">Universal Model Routing for Efficient LLM Inference</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-32" href="#footnote-anchor-32" class="footnote-number" contenteditable="false" target="_self">32</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=4ST2YyTjI7">LD-MoLE, "Learnable Dynamic Routing for Mixture of LoRA Experts"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-33" href="#footnote-anchor-33" class="footnote-number" contenteditable="false" target="_self">33</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=ei1bRG971A">DND, "Boosting Large Language Models with Dynamic Nested Depth"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-34" href="#footnote-anchor-34" class="footnote-number" contenteditable="false" target="_self">34</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=0mNnINd2z5">"Strategic Scaling of Test-Time Compute: A Bandit Learning Approach"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-35" href="#footnote-anchor-35" class="footnote-number" contenteditable="false" target="_self">35</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=zF0A0xw3HZ">vCache, "Verified Semantic Prompt Caching"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-36" href="#footnote-anchor-36" class="footnote-number" contenteditable="false" target="_self">36</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39064590/architecting-controllable-parametric-memory-in-language-models">Aditi Raghunathan, "Architecting Controllable Parametric Memory in Language Models," ICLR 2026 MemAgents workshop, SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-37" href="#footnote-anchor-37" class="footnote-number" contenteditable="false" target="_self">37</a><div class="footnote-content"><p><strong><a href="https://sites.google.com/view/memagent-iclr26/schedule">ICLR 2026 MemAgents workshop schedule</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-38" href="#footnote-anchor-38" class="footnote-number" contenteditable="false" target="_self">38</a><div class="footnote-content"><p>Conference source-pack evidence from the ICLR 2026 analysis pipeline, summarized in <a href="https://drive.google.com/file/d/1mqrmUDXS4ZW11UHqu83mu1Y5XSMbMgP5/view">09_efficiency_systems_architectures.md</a>, <a href="https://drive.google.com/file/d/1v2_jiA_pY1_Jj-pH9ycV-GWm5_YuulPm/view">09_efficiency_systems_architectures__papers.md</a>, and <a href="https://drive.google.com/file/d/11uL4W3a0r_M4m95RWnEQEoJ7nvh1STdp/view">09_efficiency_systems_architectures__transcripts.md</a> (Google Drive): the efficiency/systems reconciliation frames the area as adaptive full-stack co-design across tokens, layers, experts, KV entries, precision formats, routes, and reasoning steps; the exhaustive paper pass flags draft-guided KV dropping, cross-layer KV reconstruction, PagedAttention-style cache management, verified sparse attention, dynamic routing, and dynamic depth as recurring mechanisms. Public anchors are the ICLR OpenReview papers cited in Trend 1, plus vCache<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-63" href="#footnote-63" target="_self">63</a>, Universal Model Routing<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-64" href="#footnote-64" target="_self">64</a>, Race Attention<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-65" href="#footnote-65" target="_self">65</a>, and the MemAgents workshop<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-66" href="#footnote-66" target="_self">66</a>.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-39" href="#footnote-anchor-39" class="footnote-number" contenteditable="false" target="_self">39</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=6iRZvJiC9Q">OpenCUA, "Open Foundations for Computer-Use Agents"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-40" href="#footnote-anchor-40" class="footnote-number" contenteditable="false" target="_self">40</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=LZnKNApvhG">TheAgentCompany, "Benchmarking LLM Agents on Consequential Real World Tasks"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-41" href="#footnote-anchor-41" class="footnote-number" contenteditable="false" target="_self">41</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=FLiMxTkIeu">AGENTIF, "Benchmarking Large Language Models Instruction Following Ability in Agentic Scenarios"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-42" href="#footnote-anchor-42" class="footnote-number" contenteditable="false" target="_self">42</a><div class="footnote-content"><p><strong><a href="https://neurips.cc/virtual/2025/social/129335">NeurIPS 2025 Social, "Evaluating Agentic Systems: Bridging Research Benchmarks and Real-World Impact"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-43" href="#footnote-anchor-43" class="footnote-number" contenteditable="false" target="_self">43</a><div class="footnote-content"><p>Conference transcript evidence from AI4NextG @ NeurIPS'25, summarized in the conference analysis reports <a href="https://drive.google.com/file/d/1pnw4MgiPx5hLAtAqjzbSc4hMjHxEXFXd/view">18_agentic_ai_autonomous_systems.md</a> and <a href="https://drive.google.com/file/d/1ZfDR_dF_CcXLDahyVSaBVyT7OguZn__x/view">19_notable_quotes.md</a> (Google Drive): slide 21 at 46:40 describes task-level agents with goals, memory, feedback, tool/action selection, and replanning; timestamp 06:47:13 describes agentic AI as non-deterministic and requiring a different QA process. Public conference anchors are the AI4NextG NeurIPS 2025 SlidesLive collection<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-67" href="#footnote-67" target="_self">67</a> and AI4NextG workshop site<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-68" href="#footnote-68" target="_self">68</a>.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-44" href="#footnote-anchor-44" class="footnote-number" contenteditable="false" target="_self">44</a><div class="footnote-content"><p><strong><a href="https://agentwild-workshop.github.io/">ICLR 2026 Agents in the Wild workshop schedule</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-45" href="#footnote-anchor-45" class="footnote-number" contenteditable="false" target="_self">45</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39064314/from-llms-to-agents-the-evaluation-challenge">Bing Liu, "From LLMs to Agents: The Evaluation Challenge," ICLR 2026 Agents in the Wild workshop, SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-46" href="#footnote-anchor-46" class="footnote-number" contenteditable="false" target="_self">46</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39064323/compound-ai-systems-design-patterns-for-secure-multiagent-deployments">Jared Quincy Davis, "Compound AI Systems: Design Patterns for Secure Multi-Agent Deployments," ICLR 2026 Agents in the Wild workshop, SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-47" href="#footnote-anchor-47" class="footnote-number" contenteditable="false" target="_self">47</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/39063831/the-art-and-science-of-benchmarking-agents">Fred Sala, "The Art and Science of Benchmarking Agents," ICLR 2026 DATA-FM workshop, SlidesLive</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-48" href="#footnote-anchor-48" class="footnote-number" contenteditable="false" target="_self">48</a><div class="footnote-content"><p>Conference transcript and source-pack evidence from the ICLR 2026 analysis pipeline, summarized in <a href="https://drive.google.com/file/d/1IayDpom44e4TRHBhRsxrfZLNu4ZYFKU0/view">trend_report.md</a>, <a href="https://drive.google.com/file/d/1PvW0kXzCw70_0-CsmQH1EOpqRdGRShCn/view">executive_summary.md</a>, <a href="https://drive.google.com/file/d/1QMtWzXk7Ij7t02hzy-zX-Ao_bnWQrFHL/view">01_reasoning_post_training.md</a>, <a href="https://drive.google.com/file/d/1je57xzY1QzaJz_Rek7f-7KTHkylqOYmC/view">02_agentic_systems.md</a>, and the <a href="https://drive.google.com/file/d/1KpezKSQ7XrCJ_nahl3z4OduMmysEveYB/view">02_agentic_systems topic reconciliation</a> (Google Drive): the trend report frames the conference-wide shift from isolated model products to composed systems; the executive summary makes systems the unit of progress and risk; the reasoning report frames reasoning as a control/system problem; and the agentic-systems report treats agents as persistent action loops over tools, memory, GUIs, browsers, APIs, files, simulators, and other agents. Public conference anchors are the AIWILD workshop<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-69" href="#footnote-69" target="_self">69</a>, Bing Liu SlidesLive talk<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-70" href="#footnote-70" target="_self">70</a>, Fred Sala SlidesLive talk<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-71" href="#footnote-71" target="_self">71</a>, Agentic Context Engineering<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-72" href="#footnote-72" target="_self">72</a>, and the MemAgents workshop<a class="footnote-anchor" data-component-name="FootnoteAnchorToDOM" id="footnote-anchor-73" href="#footnote-73" target="_self">73</a>.</p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-49" href="#footnote-anchor-49" class="footnote-number" contenteditable="false" target="_self">49</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=eC4ygDs02R">Agentic Context Engineering, "Evolving Contexts for Self-Improving Language Models"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-50" href="#footnote-anchor-50" class="footnote-number" contenteditable="false" target="_self">50</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=a7Qa4CcHak">Terminal-Bench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-51" href="#footnote-anchor-51" class="footnote-number" contenteditable="false" target="_self">51</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=41xrZ3uGuI">FeatureBench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-52" href="#footnote-anchor-52" class="footnote-number" contenteditable="false" target="_self">52</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=7cHhcrbr6x">WebDS</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-53" href="#footnote-anchor-53" class="footnote-number" contenteditable="false" target="_self">53</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=3x4SDbXbgl">Computer Agent Arena</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-54" href="#footnote-anchor-54" class="footnote-number" contenteditable="false" target="_self">54</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=2H03gm4Rq6">TRACE, "Towards Self-Evolving Agent Benchmarks: Validatable Agent Trajectory via Test-Time Exploration"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-55" href="#footnote-anchor-55" class="footnote-number" contenteditable="false" target="_self">55</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=vUaY1t64ZZ">Holistic Agent Leaderboard, "The Missing Infrastructure for AI Agent Evaluation"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-56" href="#footnote-anchor-56" class="footnote-number" contenteditable="false" target="_self">56</a><div class="footnote-content"><p><strong><a href="https://openreview.net/forum?id=7XYjeL46co">MCP-SafetyBench</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-57" href="#footnote-anchor-57" class="footnote-number" contenteditable="false" target="_self">57</a><div class="footnote-content"><p><strong><a href="https://lock-llm.github.io/">Lock-LLM NeurIPS 2025 workshop site</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-58" href="#footnote-anchor-58" class="footnote-number" contenteditable="false" target="_self">58</a><div class="footnote-content"><p><strong><a href="https://slideslive.com/neurips-2025/workshop-ai-and-ml-for-nextgeneration-wireless-communications-and-networking-ai4nextg-neurips25">NeurIPS 2025 SlidesLive collection, "Workshop: AI and ML for Next-Generation Wireless Communications and Networking (AI4NextG @ NeurIPS'25)"</a></strong></p></div></div><div class="footnote" data-component-name="FootnoteToDOM"><a id="footnote-59" href="#footnote-anchor-59" class="footnote-number" contenteditable="false" target="_self">59</a><div class="footnote-content"><p><strong><a href="https://ai4nextg.github.io/">AI4NextG workshop site, previous edition at NeurIPS 2025</a></strong></p></div></div>]]></content:encoded></item></channel></rss>