[{"data":1,"prerenderedAt":1154},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Fplanner-in-the-loop":321,"\u002Fnews\u002Finsights\u002Fplanner-in-the-loop-surround":632},[4,8,17,21,25,29,33,309,313,317],{"title":5,"path":6,"stem":7},"About","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36,"children":37},"News","\u002Fnews","news",[38,41,65],{"title":39,"path":35,"stem":40},"News & Insights","news\u002Findex",{"title":18,"path":42,"stem":43,"children":44},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[45,49,53,57,61],{"title":46,"path":47,"stem":48},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":50,"path":51,"stem":52},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":54,"path":55,"stem":56},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":58,"path":59,"stem":60},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":62,"path":63,"stem":64},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":66,"stem":67,"children":68},"\u002Fnews\u002Finsights","news\u002Finsights",[69,73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297,301,305],{"title":70,"path":71,"stem":72},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":74,"path":75,"stem":76},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":78,"path":79,"stem":80},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":82,"path":83,"stem":84},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":86,"path":87,"stem":88},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":90,"path":91,"stem":92},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":94,"path":95,"stem":96},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":98,"path":99,"stem":100},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":102,"path":103,"stem":104},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":106,"path":107,"stem":108},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":110,"path":111,"stem":112},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":114,"path":115,"stem":116},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":118,"path":119,"stem":120},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":122,"path":123,"stem":124},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":126,"path":127,"stem":128},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":130,"path":131,"stem":132},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":134,"path":135,"stem":136},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":138,"path":139,"stem":140},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":142,"path":143,"stem":144},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":146,"path":147,"stem":148},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":150,"path":151,"stem":152},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":154,"path":155,"stem":156},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":158,"path":159,"stem":160},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":162,"path":163,"stem":164},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":166,"path":167,"stem":168},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":170,"path":171,"stem":172},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":174,"path":175,"stem":176},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":178,"path":179,"stem":180},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":182,"path":183,"stem":184},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":186,"path":187,"stem":188},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":190,"path":191,"stem":192},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":194,"path":195,"stem":196},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":198,"path":199,"stem":200},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":202,"path":203,"stem":204},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":206,"path":207,"stem":208},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":210,"path":211,"stem":212},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":214,"path":215,"stem":216},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":218,"path":219,"stem":220},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":222,"path":223,"stem":224},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":226,"path":227,"stem":228},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":230,"path":231,"stem":232},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":234,"path":235,"stem":236},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":238,"path":239,"stem":240},"Orchestration Without a Signal From the Room","\u002Fnews\u002Finsights\u002Fplanner-in-the-loop","news\u002Finsights\u002Fplanner-in-the-loop",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Body Was Supposed to Be the Easy Part","\u002Fnews\u002Finsights\u002Frobot-embodiment-interface","news\u002Finsights\u002Frobot-embodiment-interface",{"title":270,"path":271,"stem":272},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":274,"path":275,"stem":276},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":278,"path":279,"stem":280},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":282,"path":283,"stem":284},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":286,"path":287,"stem":288},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":290,"path":291,"stem":292},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":294,"path":295,"stem":296},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":298,"path":299,"stem":300},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":302,"path":303,"stem":304},"Algorithms Used in Autonomous Fighter Jet Flight Research","\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy","news\u002Finsights\u002Fvenom-f16-flight-autonomy",{"title":306,"path":307,"stem":308},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":310,"path":311,"stem":312},"Privacy Policy","\u002Fprivacy","privacy",{"title":314,"path":315,"stem":316},"Research","\u002Fresearch","research",{"title":318,"path":319,"stem":320},"Terms of Service","\u002Fterms","terms",{"id":322,"title":238,"body":323,"date":613,"description":614,"extension":615,"image":616,"meta":617,"navigation":629,"path":239,"seo":630,"stem":240,"__hash__":631},"insights\u002Fnews\u002Finsights\u002Fplanner-in-the-loop.md",{"type":324,"value":325,"toc":598},"minimark",[326,345,355,359,362,365,369,381,398,401,405,408,418,426,429,433,443,453,467,481,489,493,496,506,509,512],[327,328,331,332,331,338],"div",{"className":329},[330],"page-title","\n  ",[333,334,238],"h1",{"className":335,"id":337},[336],"page-title__main","orchestration-without-a-signal-from-the-room",[339,340,344],"h2",{"className":341,"id":343},[342],"page-title__sub","agent-graphs-can-call-tools-the-conductor-often-does-not-hear-whether-the-job-finished","Agent Graphs Can Call Tools. The Conductor Often Does Not Hear Whether the Job Finished.",[327,346,348,349],{"style":347},"width: 100%; padding: 2%;","\n    ",[350,351],"img",{"src":352,"alt":353,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1745328599097-5b2adfa0087e?w=1200&auto=format&fit=crop","A conductor facing an orchestra during rehearsal, listening rather than reading a score in isolation","width: 100%; height: auto;",[356,357,358],"p",{},"A conductor who has only ever studied the score is not the same as a conductor who has stood in front of that particular orchestra. The score already knows which entrance belongs to the horns. What it cannot tell anyone is whether those horns will actually arrive on time tonight, or whether the first violin is sitting a half-beat late, or whether the room itself is swallowing the low strings. Those facts only show up in rehearsal.",[356,360,361],{},"Most production agent stacks are still closer to the score. A large language model, a system trained to generate text by predicting the next word, gets wrapped in a graph of specialists with names like planner, executor, critic, and coder. Each node is a frozen model plus a prompt. Tools get called. Memory gets appended. The graph looks like architecture. Orchestration, the work of deciding what happens next, which tool, which sub-goal, whether the last result was even usable, often does not receive a learning signal from the trajectory it just caused. The room already answered. The graph did not listen.",[356,363,364],{},"From a systems perspective, that missing wire is the whole story. A deployed controller that cannot hear the plant will keep conducting from last week's diagram. Prompt edits after a surprising failure are not the same as a gradient, the training signal that tells a model how to change. They are a human rewriting the score by hand.",[339,366,368],{"id":367},"a-score-with-every-entrance-marked","A Score With Every Entrance Marked",[356,370,371,372,380],{},"The conversation frameworks that made agent graphs easy to ship did something useful. They made orchestration programmable as talk among specialists, humans, and tools, without requiring a training loop for that particular collaboration ",[373,374,375],"sup",{},[376,377,379],"a",{"href":378},"#source-1","[1]",". The graph became a product surface. Roles could be named. Handoffs could be coded. What those designs did not attach, by construction, was a learning signal to the talk itself. When a search result was noisy, a code tool returned an exception, or an early sub-goal was the wrong one, the usual fix was another prompt. Prompting problems tend to be rewritten after every surprising failure.",[327,382,348,384,348,388,348,394],{"style":383},"width: 100%; margin: 20px 0;",[350,385],{"src":386,"alt":387,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1750449821191-68a5facd2be7?w=1200&auto=format&fit=crop","A vintage control room of labeled switches and analog panels that stay in the same positions",[389,390,393],"h3",{"style":391,"id":392},"margin: 1rem 0 0.5rem 0;","every-switch-already-has-a-label","Every Switch Already Has a Label",[356,395,397],{"style":396},"margin: 0;","A room like this looks finished. Every circuit has a nameplate and a home position. What it does not have is a way for the person at the board to get better at using it from the last shift's near misses. Frozen agent graphs have the same finished look. The planner, executor, and verifier all have roles. In the default design, none of those roles takes a gradient from what the last tool call actually returned.",[356,399,400],{},"That is orchestration without a signal from the room. It can still run a job. It cannot, on its own, get better at sequencing the job from the job's own outcome.",[339,402,404],{"id":403},"a-transcript-is-not-the-room-either","A Transcript Is Not the Room Either",[356,406,407],{},"The obvious patch is to train the orchestrator after all, just not in the live loop. Collect traces from a stronger model, copy them token by token, and call the result learning. The traces look like rehearsal notes. They are still a score. They were written for a different night, a different band, a different set of missed entrances.",[356,409,410,411,417],{},"Work on in-the-flow agent training put that patch to a direct test. Inside the same modular graph, a planner, an executor, a verifier, and a generator sharing a structured memory, distilling a stronger model's planner traces into a smaller planner produced a collapse, a 19 percent drop in average accuracy relative to leaving that planner frozen ",[373,412,413],{},[376,414,416],{"href":415},"#source-2","[2]",". The imitation objective was answering the wrong question. It rewarded looking like a good transcript, not finishing the work after a tool had already gone sideways.",[356,419,420,421,425],{},"The same graph, with the planner updated from the live rollouts it actually induced, recovered that ground. On-policy training, updates from the trajectories the current system produces rather than from a cleaned-up archive, is what puts the orchestrator back in the room. The planner still sees the live state, the query, the toolset, the current memory, because that is the state it will face at inference. After that training, tool choice moved with the task instead of staying at a default search habit. Broad factual questions pulled more web search. A medical question set spent more of the budget inside Wikipedia and page-level retrieval ",[373,422,423],{},[376,424,416],{"href":415},". A frozen prompt is supposed to produce that kind of shift and often does not, because the prompt cannot see, in gradient terms, which of those choices ended in a correct answer.",[356,427,428],{},"The lesson is narrower than \"train a planner.\" It is about which signal counts. A transcript is a record of how someone else conducted a different night. A live outcome is what this orchestra, in this room, actually did.",[339,430,432],{"id":431},"dumping-the-outcome-through-the-whole-band","Dumping the Outcome Through the Whole Band",[356,434,435,436,442],{},"Some stacks do send a final correct-or-incorrect score back through the model that issued the tool calls. The search engine sits in the environment. The model learns when to query and how to use what comes back, with retrieved tokens masked so the update does not try to credit the search engine's text as if the model had written it ",[373,437,438],{},[376,439,441],{"href":440},"#source-3","[3]",". That is a real signal. It is also a signal dumped through every thought, every tool choice, and every wording decision in one long context.",[356,444,445,446,452],{},"The room is noisy. Tool output is often off the pretrained distribution. Even when those tokens are masked from the loss, the model's next generation inherits the shift, samples increasingly unlikely tokens, and the gradient can explode. One practical response has been to throw out entire trajectories that contain a void turn, a response that produced neither a code block nor a final answer, because those turns were pulling the update in the wrong direction ",[373,447,448],{},[376,449,451],{"href":450},"#source-4","[4]",". The filter is a way of saying the room spoke, and some of what it said was not a usable learning signal.",[356,454,455,456,462,463,466],{},"A related failure shows up when the only reward is that the episode worked. Reward variability collapses, gradients spike, and the agent starts repeating locally rewarded patterns that do not amount to reasoning. Shallow strategies and hallucinated thoughts are easy to grow if the signal has nothing to say about whether the thoughts were real ",[373,457,458],{},[376,459,461],{"href":460},"#source-5","[5]",". That sits next to an ",[376,464,465],{"href":259},"earlier Thinkata look at temporal credit assignment",". A single terminal score is a hard thing to send backward through a long chain. Stuffing it through the entire policy makes the chain even longer.",[327,468,348,469,348,474,348,478],{"style":383},[350,470],{"src":471,"alt":472,"style":473},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1775519519950-9c48495b5488?w=1200&h=1200&fit=crop&crop=focalpoint&fp-x=0.52&fp-y=0.38&auto=format","An industrial panel of gauges and switches where only some circuits are live","width: 100%; aspect-ratio: 1 \u002F 1; object-fit: cover;",[389,475,477],{"style":391,"id":476},"the-wire-has-to-reach-the-loop-that-chooses","The Wire Has to Reach the Loop That Chooses",[356,479,480],{"style":396},"A working plant does not rewire every gauge because one loop is drifting. It retunes the controller that actually sets the next action. The learning signal has to land somewhere specific. Spread it across every token in a growing context and the noise in the room can swamp the update. Leave it unconnected and the labeled panel does not learn from the last shift.",[356,482,483,484,488],{},"The engineering question is not whether a terminal outcome is too crude. Outcome rewards are often the only score that can be checked. The question is where that crude score is allowed to change the system. One measured answer has been to keep the modular graph and copy a single verifiable result, right or wrong at the end, onto every orchestrator turn, while each turn still conditions on the full memory so far ",[373,485,486],{},[376,487,416],{"href":415},". The local decision is not blind. The update still points at whether the job finished. That is a systems choice about which module is allowed to hear the room, not a new theory of credit.",[339,490,492],{"id":491},"how-much-of-the-room-needs-to-hear-it","How Much of the Room Needs to Hear It",[356,494,495],{},"Training only the conductor is a bet about the rest of the stack. Tools, executors, and verifiers can stay versioned infrastructure. When the toolset or the task mix changes, the thing that gets retrained is the loop that chooses, not the whole orchestra. That matches how a lot of production software is already pinned.",[356,497,498,499,505],{},"It is also incomplete in a specific way. If the models that actually run the steps stay frozen while only the designer or planner learns, part of the loop is still conducting from a score. Work on automatic multi-agent systems has started treating that gap as a ceiling. Jointly training the side that writes the workflow and the side that executes it produced gains that leaving either side frozen did not, and the two roles appeared to improve on a staggered schedule rather than all at once ",[373,500,501],{},[376,502,504],{"href":503},"#source-6","[6]",".",[356,507,508],{},"That does not settle how much of an agent graph should be trainable. For a team whose executor is a deterministic API, a search endpoint, a code runner, a pinned verifier, putting the learning signal on the orchestrator is the operable experiment. For a team whose executors are themselves language models with their own failure modes, freezing them may be the part of the room that later looks deaf. Domain practitioners will know better than a systems argument whether a given executor is a tool or a second conductor.",[356,510,511],{},"What does seem worth taking seriously is the missing wire itself. A prompt-orchestrated graph is a practical way to run specialists. It does not, by itself, teach the module that sequences them. Sending the outcome through every token of a monolithic policy can teach that sequencing, and then spend the training budget on not exploding. The option that matches the rest of a production stack is smaller. Pin the tools. Pin the memory format. Attach a learning signal to the loop that is actually conducting, from the trajectory that loop just caused, in the room it is standing in tonight.",[327,513,331,517,331,520],{"className":514},[515,516],"references","mt-8",[339,518,519],{"id":515},"References",[521,522,348,528,348,546,348,557,348,567,348,577,348,587,331],"ol",{"className":523},[524,525,526,527],"list-decimal","list-inside","space-y-2","mt-4",[529,530,532,533,537,538],"li",{"id":531},"source-1","Q. Wu et al., \"AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation,\" ",[534,535,536],"em",{},"arXiv",", 2023, ",[376,539,545],{"href":540,"target":541,"className":542},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2308.08155","_blank",[543,544],"text-blue-600","underline","[Online]",[529,547,549,550,552,553],{"id":548},"source-2","Z. Li et al., \"In-the-Flow Agentic System Optimization for Effective Planning and Tool Use,\" ",[534,551,536],{},", 2025, ",[376,554,545],{"href":555,"target":541,"className":556},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.05592",[543,544],[529,558,560,561,552,563],{"id":559},"source-3","B. Jin et al., \"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning,\" ",[534,562,536],{},[376,564,545],{"href":565,"target":541,"className":566},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2503.09516",[543,544],[529,568,570,571,552,573],{"id":569},"source-4","Z. Xue et al., \"SimpleTIR: End-to-End Reinforcement Learning for Multi-Turn Tool-Integrated Reasoning,\" ",[534,572,536],{},[376,574,545],{"href":575,"target":541,"className":576},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.02479",[543,544],[529,578,580,581,552,583],{"id":579},"source-5","Z. Wang et al., \"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning,\" ",[534,582,536],{},[376,584,545],{"href":585,"target":541,"className":586},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2504.20073",[543,544],[529,588,590,591,593,594],{"id":589},"source-6","Y. Zhang et al., \"MetaAgent-X: Breaking the Ceiling of Automatic Multi-Agent Systems via End-to-End Reinforcement Learning,\" ",[534,592,536],{},", 2026, ",[376,595,545],{"href":596,"target":541,"className":597},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.14212",[543,544],{"title":599,"searchDepth":600,"depth":600,"links":601},"",2,[602,603,607,608,611,612],{"id":343,"depth":600,"text":344},{"id":367,"depth":600,"text":368,"children":604},[605],{"id":392,"depth":606,"text":393},3,{"id":403,"depth":600,"text":404},{"id":431,"depth":600,"text":432,"children":609},[610],{"id":476,"depth":606,"text":477},{"id":491,"depth":600,"text":492},{"id":515,"depth":600,"text":519},"2026-09-07","Agent stacks can look like architecture, specialists, tools, memory, a planner, and still send no training signal back to the module that is conducting the work. The room already answered. The graph often does not listen.","md",{"src":352},{"authors":618,"badge":624,"source":626},[619],{"avatar":620,"name":622,"to":623},{"src":621},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":625},"Agent Systems",{"name":627,"url":628},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":238,"description":614},"0Kpu6a_IGPAxywvnv3uk3v9OZc-Tg293WrjvZc_kotI",[633,830],{"id":634,"title":266,"body":635,"date":818,"description":819,"extension":615,"image":820,"meta":821,"navigation":629,"path":267,"seo":828,"stem":268,"__hash__":829,"_path":267},"insights\u002Fnews\u002Finsights\u002Frobot-embodiment-interface.md",{"type":324,"value":636,"toc":808},[637,649,655,658,661,665,681,684,712,716,728,731,735,743,747,750],[327,638,331,640,331,644],{"className":639},[330],[333,641,266],{"className":642,"id":643},[336],"the-body-was-supposed-to-be-the-easy-part",[339,645,648],{"className":646,"id":647},[342],"every-humanoid-robot-has-needed-its-own-policy-trained-from-scratch-recent-research-asks-whether-a-shared-interface-can-finally-change-that","Every Humanoid Robot Has Needed Its Own Policy, Trained From Scratch. Recent Research Asks Whether a Shared Interface Can Finally Change That.",[327,650,348,651],{"style":347},[350,652],{"src":653,"alt":654,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1769839271832-cfd7a1f6854f?w=1200&h=675&fit=crop&crop=focalpoint&fp-x=0.5&fp-y=0.32&auto=format","A humanoid robot's head, neck, and shoulders, wearing a blue lanyard",[356,656,657],{},"A humanoid robot from one company does not share a skeleton with one from another. Arm lengths differ. Joint counts differ. Torque limits, sensor placement, and even how a knee is allowed to bend differ. For most of robotics' history, none of that has mattered much to the software riding on top, because a control policy, the trained model that decides what a robot's joints should do next, has been built for one specific machine and has stayed there. Retrain it, or even slightly change the robot's arm length or joint count, and the policy is often useless. It behaves less like software and more like a badge that only opens one particular door, no matter how many identical-looking doors get added down the hall.",[356,659,660],{},"That is the quieter problem sitting underneath the current wave of humanoid robot announcements. Several companies now ship their own distinct design, and a policy trained to walk one company's robot cannot walk another's, no matter how similar the two machines look standing still. What changed recently is a cluster of papers, working from different starting points, converging on the same fix, treating a robot's specific body the way a computer treats a specific chip, as a detail a shared interface should be able to hide rather than something a policy has to relearn from nothing.",[339,662,664],{"id":663},"one-body-learned-well","One Body, Learned Well",[356,666,667,668,671,672,676,677,505],{},"The first problem was more basic than making a policy work across many robots. It was making a single policy work well on one. Humanoid control has traditionally meant a library of separate, narrow skills, a walking gait here, a reaching motion there, each trained and tuned on its own. NVIDIA's SONIC controller, described in a paper published in ",[534,669,670],{},"Science Robotics",", took a different approach, training one controller with roughly forty million parameters on a motion capture library far larger than earlier humanoid controllers had used, teaching it a broad vocabulary of human movement rather than a fixed set of tricks ",[373,673,674],{},[376,675,379],{"href":378},". The result generalizes to movements it never specifically trained on, and it accepts commands from several different sources, virtual reality teleoperation, recorded video, or a separate vision language model, through what its authors describe as a single shared token space ",[373,678,679],{},[376,680,379],{"href":378},[356,682,683],{},"That shared token space is worth sitting with. It is not just a bigger model. It is a common vocabulary that different kinds of commands, whether from a human in a headset or another AI system giving instructions, get translated into before the robot's body ever has to interpret them. In computing terms, it functions like an instruction set, the fixed vocabulary of operations a chip agrees to support so that software written once can run without knowing exactly which transistors will execute it. SONIC built that instruction set for one specific robot body. The question left open was whether the same idea would survive contact with a different body entirely.",[327,685,348,686,348,690,348,694],{"style":383},[350,687],{"src":688,"alt":689,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1564517945244-d371c925640b?w=1200&auto=format&fit=crop","Two different electrical adapters lying side by side on a plain background",[389,691,693],{"style":391,"id":692},"any-body-same-instruction-set","Any Body, Same Instruction Set",[356,695,696,697,701,702,706,707,711],{"style":396},"Swap the socket and the appliance still runs, because the plug was never the part doing the work. Robotics had already tested that logic once, on robot arms rather than legs, when the Open X-Embodiment collaboration pooled data from twenty-two different robots, contributed by research labs across several countries, and found that a policy trained on all of them transferred usefully to individual machines it had never specifically seen ",[373,698,699],{},[376,700,416],{"href":415},". Two more recent efforts pushed that same logic into humanoid whole-body control specifically. One system, built by researchers at Shanghai Jiao Tong University and the Shanghai Artificial Intelligence Laboratory, trains on a wide distribution of simulated robot shapes and physical properties so the resulting policy generalizes to real humanoids it never trained on directly, without any robot-specific retraining ",[373,703,704],{},[376,705,441],{"href":440},". A separate, newer framework goes a step further, explicitly separating the parts of a motion that come from shared human movement semantics, the timing and structure any body might share, from the parts that are specific to one robot's own physical execution, then routing each through a different piece of the model ",[373,708,709],{},[376,710,451],{"href":450},". Different research groups, different robot fleets, and the same underlying decision, that a body's particular geometry should sit behind the interface rather than baked into it.",[339,713,715],{"id":714},"what-the-interface-is-actually-made-of","What the Interface Is Actually Made Of",[356,717,718,719,723,724,505],{},"Saying a robot's body sits behind an interface is not yet an explanation. An interface only earns that name if something on one side of it stays fixed while something on the other side is free to change, and the harder design question in this line of research is exactly where that line gets drawn. The Shanghai team draws it before its model ever meets a specific robot, training the shared representation against a wide spread of simulated bodies and physical properties so that no single robot's proportions get baked into what the policy has learned to expect, while keeping the meaning of a given sensor reading or motor command consistent no matter which robot happens to be listening ",[373,720,721],{},[376,722,441],{"href":440},". The newer cross-embodiment framework draws the same line in a different place, splitting a human-centered vocabulary, tokens that describe a motion's shape and timing regardless of who or what is performing it, from a set of small, robot-specific modules whose only job is translating that shared vocabulary into one particular robot's own joints and proprioception ",[373,725,726],{},[376,727,451],{"href":450},[356,729,730],{},"What both designs have in common matters more than where they differ. The part that has to be rebuilt for a new robot is deliberately kept small, and almost everything the policy actually knows about movement lives on the shared side of that line. That allocation is the entire point of building an interface in the first place. A translation layer that has to relearn most of what it knows every time a new robot shows up is not really an interface, it is just a second training run with extra steps.",[339,732,734],{"id":733},"what-the-interface-does-not-settle","What the Interface Does Not Settle",[356,736,737,738,742],{},"None of this closes the distance between a research demonstration and a robot working somewhere messier than a motion capture stage or a lab floor, and that gap is worth stating plainly rather than waving past. Even a single, well-behaved robot body running in a clean simulator does not automatically match how that same body moves once it is welded, motored, and standing on real ground, a mismatch specific enough that an earlier project needed a dedicated second training stage just to correct for it after the fact ",[373,739,740],{},[376,741,461],{"href":460},". Extending that correction from one lab robot to a policy meant to generalize across many bodies, then further out to homes, warehouses, or outdoor terrain nobody trained on, is a considerably larger claim than anything in the papers above actually tested. What the interface research settles is narrower and, in its own way, more useful, whether a body has to be relearned from scratch every time it changes shape.",[339,744,746],{"id":745},"what-this-suggests","What This Suggests",[356,748,749],{},"Every new humanoid robot announcement invites the same question, whether this particular machine can actually do the job asked of it. The research underneath that question suggests the frontier is not really about any one body. It is about whether a shared interface, a vocabulary of motion that means the same thing regardless of which machine is listening, can keep holding as more and increasingly different bodies get added behind it. Computing went through a comparable split decades ago, separating what a chip is from what a program written for it has to know, and that separation took a long, uneven stretch of competing designs before it became something the field simply assumed rather than argued about. Robotics looks to be partway through an equivalent shift now, still testing how far one interface can stretch before a new robot's proportions or actuators break the abstraction, and how much of the underlying skill survives that stretch intact.",[327,751,331,753,331,755],{"className":752},[515,516],[339,754,519],{"id":515},[521,756,348,758,348,768,348,779,348,788,348,797,331],{"className":757},[524,525,526,527],[529,759,760,761,763,764],{"id":531},"Z. Luo et al., \"Supersizing Motion Tracking for Natural Humanoid Whole-Body Control,\" ",[534,762,670],{},", vol. 11, no. 117, eaed4592, 2026. DOI: ",[376,765,545],{"href":766,"target":541,"className":767},"https:\u002F\u002Fdoi.org\u002F10.1126\u002Fscirobotics.aed4592",[543,544],[529,769,770,771,774,775],{"id":548},"A. O'Neill et al., \"Open X-Embodiment: Robotic Learning Datasets and RT-X Models,\" in ",[534,772,773],{},"Proc. IEEE International Conference on Robotics and Automation (ICRA)",", 2024, pp. 6892-6903. DOI: ",[376,776,545],{"href":777,"target":541,"className":778},"https:\u002F\u002Fdoi.org\u002F10.1109\u002FICRA57147.2024.10611477",[543,544],[529,780,781,782,593,784],{"id":559},"Y. Xue et al., \"Scalable and General Whole-Body Control for Cross-Humanoid Locomotion,\" ",[534,783,536],{},[376,785,545],{"href":786,"target":541,"className":787},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2602.05791",[543,544],[529,789,790,791,593,793],{"id":569},"J. Zhang et al., \"X-WBC: A Cross-Embodiment Foundation Model for Humanoid Whole-Body Control,\" ",[534,792,536],{},[376,794,545],{"href":795,"target":541,"className":796},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.15213",[543,544],[529,798,799,800,803,804],{"id":579},"T. He et al., \"ASAP: Aligning Simulation and Real-World Physics for Learning Agile Humanoid Whole-Body Skills,\" in ",[534,801,802],{},"Proc. Robotics: Science and Systems (RSS)",", 2025. DOI: ",[376,805,545],{"href":806,"target":541,"className":807},"https:\u002F\u002Fdoi.org\u002F10.15607\u002FRSS.2025.XXI.066",[543,544],{"title":599,"searchDepth":600,"depth":600,"links":809},[810,811,814,815,816,817],{"id":647,"depth":600,"text":648},{"id":663,"depth":600,"text":664,"children":812},[813],{"id":692,"depth":606,"text":693},{"id":714,"depth":600,"text":715},{"id":733,"depth":600,"text":734},{"id":745,"depth":600,"text":746},{"id":515,"depth":600,"text":519},"2026-09-14","Humanoid robots from different companies do not share a skeleton, and until recently that meant a control policy built for one robot could not walk another. A cluster of new research treats a robot's body as a detail a shared interface should hide, not something a policy has to relearn from scratch every time it changes shape.",{"src":653},{"authors":822,"badge":825,"source":827},[823],{"avatar":824,"name":622,"to":623},{"src":621},{"label":826},"Embodied AI",{"name":627,"url":628},{"title":266,"description":819},"ksvAsVQopvVcFzVTMVGHQNVNPM9rKnmRVRvmK7p0RPk",{"id":831,"title":302,"body":832,"date":1142,"description":1143,"extension":615,"image":1144,"meta":1145,"navigation":629,"path":303,"seo":1152,"stem":304,"__hash__":1153,"_path":303},"insights\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy.md",{"type":324,"value":833,"toc":1131},[834,846,852,871,874,878,890,902,914,918,925,942,964,967,971,978,1000,1004,1021,1029,1045,1049,1052],[327,835,331,837,331,841],{"className":836},[330],[333,838,302],{"className":839,"id":840},[336],"algorithms-used-in-autonomous-fighter-jet-flight-research",[339,842,845],{"className":843,"id":844},[342],"hierarchical-reinforcement-learning-and-liquid-time-constant-networks-in-light-of-darpas-venom-program","Hierarchical Reinforcement Learning and Liquid Time-Constant Networks, in Light of DARPA's VENOM Program",[327,847,348,848],{"style":347},[350,849],{"src":850,"alt":851,"style":354},"\u002Fimg\u002Ff16-fighter-jet-xy4g0V6dZEc-unsplash.jpg","An F-16 fighter jet banking in flight against a clear sky",[356,853,854,855,859,860,864,865,505],{},"In July 2026, DARPA and the U.S. Air Force announced that a modified F-16 had flown under the control of an \"AI agent,\" part of a program called the Viper Experimentation and Next-generation Operations Model, or VENOM ",[373,856,857],{},[376,858,379],{"href":378},". The release describes a hardware and software kit bolted onto an otherwise unmodified F-16, letting a pilot flip a switch between human and AI control ",[373,861,862],{},[376,863,379],{"href":378},". At least four F-16s have been outfitted with the kit so far, including an added auto-throttle that lets the AI agent regulate thrust alongside the control surfaces ",[373,866,867],{},[376,868,870],{"href":869},"#source-7","[7]",[356,872,873],{},"That headline is a good prompt to ask a broader question: what does the published research on teaching a model to fly a fighter jet, without a human in direct control, actually look like? VENOM sits downstream of a well documented lineage, DARPA's Air Combat Evolution program, and that lineage has produced two distinct, peer-reviewed approaches to that problem. Neither one is exotic by the standards of applied machine learning, and both are worth understanding in their own right.",[339,875,877],{"id":876},"the-dogfighting-brain-trust","The Dogfighting Brain Trust",[356,879,880,881,885,886,505],{},"The more heavily documented of the two threads traces back to DARPA's AlphaDogfight Trials, a 2020 competition designed to test whether an AI agent could hold its own in simulated within visual range air combat, more commonly called a dogfight. A Lockheed Martin team built an agent called PHANG-MAN that split the dogfight into three specialized sub-behaviors, a Control Zone policy for establishing a dominant position, and two variants of a shooter policy, one aggressive and one conservative about when to take a shot ",[373,882,883],{},[376,884,416],{"href":415},". A higher level \"policy selector\" watched the engagement and switched between these three specialists roughly ten times per second, weighing the current distance and angle between the jets along with the closure rate, how quickly that gap was shrinking or growing ",[373,887,888],{},[376,889,416],{"href":415},[356,891,892,893,897,898,505],{},"This is a hierarchical reinforcement learning design, meaning the system learns by trial, error, and reward rather than by copying a human, and it leans on several small, specialized learners instead of one big one. The training algorithm underneath is worth naming correctly, since it is often described elsewhere as Proximal Policy Optimization. The paper itself specifies a related but different method called Soft Actor-Critic, which rewards the system for trying a variety of moves while it is still learning instead of locking onto one tactic too early ",[373,894,895],{},[376,896,416],{"href":415},". That distinction matters more for accuracy than for the outcome, since both algorithms belong to the same toolkit that shows up across modern robotics control. What actually made PHANG-MAN notable was never really the base algorithm. It was the decomposition, training three narrow specialists and a switcher rather than asking one policy to handle every situation. In the tournament, PHANG-MAN finished second overall, and in a separate best of five match, it defeated a graduate of the U.S. Air Force's F-16 Weapons Instructor Course five wins to zero losses ",[373,899,900],{},[376,901,416],{"href":415},[356,903,904,905,909,910,505],{},"This same style of hierarchical policy work is the lineage that DARPA credits as feeding into live flights on the X-62A VISTA, a heavily modified F-16 used as the ACE program's flying testbed, which in 2024 completed the first in-air dogfight between an AI-piloted and human-piloted F-16 ",[373,906,907],{},[376,908,504],{"href":503},". VENOM's stated purpose is to take those lessons and move them off a one of a kind research aircraft and onto standard operational airframes ",[373,911,912],{},[376,913,379],{"href":378},[339,915,917],{"id":916},"liquid-time-constant-networks-a-different-approach-to-memory","Liquid Time-Constant Networks: A Different Approach to Memory",[356,919,920,921,505],{},"A separate research thread trained an entirely different kind of model to fly the same job. A team from the Department of the Air Force-MIT AI Accelerator, working with MIT Lincoln Laboratory and the university's Computer Science and Artificial Intelligence Laboratory, trained a Liquid Time-Constant network to fly the X-62A VISTA, and reportedly reached autonomous live flight in roughly six months, faster than the multi-year timelines that ACE-style reinforcement learning training has typically required ",[373,922,923],{},[376,924,441],{"href":440},[356,926,927,928,932,933,937,938,505],{},"A Liquid Time-Constant network, or LTC, takes a different approach to memory than most neural networks. A standard recurrent network, the kind used inside an LSTM, checks in on its own internal memory at fixed, evenly spaced intervals no matter what is happening around it, more like a strobe light than a dimmer switch. An LTC instead lets a differential equation, essentially a rule for how its internal state should change from one moment to the next, adjust its own update speed depending on how quickly the incoming data itself is changing ",[373,929,930],{},[376,931,451],{"href":450},". The architecture's creators also designed it to stay well behaved rather than spiral out of control as inputs keep shifting, a property standard recurrent networks are not guaranteed to have ",[373,934,935],{},[376,936,451],{"href":450},". Rather than training this network with reinforcement learning, the DAF-MIT team used imitation learning, where a model learns to copy an expert's recorded behavior directly instead of discovering a strategy on its own through trial, error, and reward ",[373,939,940],{},[376,941,441],{"href":440},[327,943,348,944,348,948,348,952],{"style":383},[350,945],{"src":946,"alt":947,"style":354},"\u002Fimg\u002Fai-chip-circuit-board-sNt81Whsncg-unsplash.jpg","A glowing AI chip embedded in a circuit board",[389,949,951],{"style":391,"id":950},"why-the-architecture-not-just-the-timeline-is-the-interesting-part","Why the Architecture, Not Just the Timeline, Is the Interesting Part",[356,953,954,955,959,960,505],{"style":396},"A chip only does what its circuit lets it do, and an LTC network's circuit is built to keep adjusting how fast it forgets. A separate, peer-reviewed study from the same MIT group tested this property directly, training LTC-based drones by imitation learning and then flying them through forests and neighborhoods they had never seen during training. Engineers have a name for that kind of mismatch between what a model saw in training and what it meets in the real world, a distribution shift ",[373,956,957],{},[376,958,461],{"href":460},". Compared against six other recurrent architectures sharing the same visual front end, the liquid networks were the ones that kept working once the scenery changed out from under them, a result the researchers attribute to the architecture learning the causal structure of the task itself, tracking the target rather than memorizing incidental details of the training environment ",[373,961,962],{},[376,963,461],{"href":460},[356,965,966],{},"That robustness to unfamiliar surroundings is not a side note. It is exactly the kind of property that matters once any flight autonomy system leaves a clean simulation for real, changing sensor conditions, something the earlier ACE-era flights on VISTA were not required to demonstrate.",[339,968,970],{"id":969},"the-same-buildup-regardless-of-algorithm","The Same Buildup, Regardless of Algorithm",[356,972,973,974,505],{},"Both approaches, and any successor to either one, share a development path, because every public account of this program family describes the same buildup. First comes software-in-the-loop simulation, running the AI purely against a simulated version of the jet. Then hardware-in-the-loop simulation, pairing that same software with real flight hardware sitting on the ground. Then constructive modeling, larger simulated exercises where computer generated aircraft stand in for real ones. Only then does live flight happen, and the whole process can run from months to years ",[373,975,976],{},[376,977,441],{"href":440},[356,979,980,981,985,986,990,991,995,996,505],{},"Much of this pipeline runs on JSBSim, an open source, C++ flight dynamics model that both the PHANG-MAN and VENOM-adjacent research use to simulate an F-16's six degrees of freedom, essentially every way the jet can move and rotate through the air, before anything touches real hardware ",[373,982,983],{},[376,984,416],{"href":415},". A newer, lighter weight tool called Tunnel, built specifically anticipating VENOM and its follow-on program's needs, wraps this kind of flight dynamics model in Gymnasium, a standard software interface used widely across reinforcement learning research, so researchers can swap in new sensors, tasks, and training methods in days rather than months ",[373,987,988],{},[376,989,441],{"href":440},". The comparison study behind Tunnel is a useful data point on its own, in a basic navigation task, an agent trained by imitation learning off a simple autopilot reliably reached its goal, while reinforcement learning agents given the same sensor data only got there some of the time ",[373,992,993],{},[376,994,441],{"href":440},". That paper also flags something closer to a design constraint than a finding, direct control of the flight surfaces by a reinforcement learning agent is generally considered unreliable, which is a likely reason ACE-era systems hand the model a higher level command, stick, rudder, and throttle position, rather than letting it move individual control surfaces on its own ",[373,997,998],{},[376,999,441],{"href":440},[339,1001,1003],{"id":1002},"why-this-keeps-getting-harder","Why This Keeps Getting Harder",[356,1005,1006,1007,1011,1012,1016,1017,505],{},"Naming these two approaches is not the same as explaining why teaching a jet to fly itself keeps getting harder as the work moves closer to a real cockpit. VISTA's ACE-era flights gave the AI agent the opponent's exact position, velocity, and orientation, pulled directly from the simulation's internal state, with no modeled sensor noise at all ",[373,1008,1009],{},[376,1010,416],{"href":415},". The published research describing this program lineage frames later stages of the work as needing to run off actual operational sensors instead, a modern scanning radar, a receiver that warns the pilot when another aircraft's radar is pointed at the jet, a broader electronic warfare system, and likely an infrared camera for spotting other aircraft without radar at all ",[373,1013,1014],{},[376,1015,441],{"href":440},". Real sensors bring real problems that a simulation's clean, ground truth data never has to answer for. A warning receiver can throw false alarms. A radar can be genuinely unsure exactly how far away or how fast something is moving. Any stretch of flight where GPS is jammed or unavailable leaves a jet's own navigation system slowly drifting off course ",[373,1018,1019],{},[376,1020,441],{"href":440},[356,1022,1023,1024,1028],{},"The same literature describes the follow-on AIR program's research goal in those terms directly, building autonomy that holds up under \"partial observability, concept drift, and uncertainty\" across scenarios involving multiple aircraft at once ",[373,1025,1026],{},[376,1027,441],{"href":440},". That is a more general statement of the same problem that hierarchical reinforcement learning and imitation learning were both built, in their own ways, to handle inside a simulator.",[1030,1031,1033,1034,1037,1039,1040],"blockquote",{"style":1032},"color: #0066CC; font-size: 1em; border-left: 4px solid #0066CC; padding-left: 1em;","\n  \"The Air Force and DARPA team has automated flight controls and sensors on a standard F-16 without changing the jet's core software. This enables an efficient pipeline for developing dominant AI for aerial combat, allowing us to rapidly innovate for the warfighter.\"",[1035,1036],"br",{},[1035,1038],{},"\n  — ",[376,1041,1044],{"href":1042,"style":1043,"target":541},"https:\u002F\u002Fwww.darpa.mil\u002Fnews\u002F2026\u002Fdarpa-us-air-force-fly-ai-controlled-f-16","color: #0066CC; text-decoration: none;","Brig. Gen. James Valpiani, DARPA",[339,1046,1048],{"id":1047},"the-bottom-line","The Bottom Line",[356,1050,1051],{},"Flight autonomy, as a field, already has two separately vetted, peer-reviewed answers to the core problem of teaching a jet to fly itself: a hierarchical reinforcement learning system that beat a human weapons instructor in simulation, and an imitation-learned continuous-time network that reached live flight faster than expected and later demonstrated a specific talent for holding up under unfamiliar conditions. Both took different paths to the same goal, and both passed through the same buildup of software simulation, hardware testing, and constructive exercises before either one ever touched a runway. That is the published toolkit behind this corner of applied machine learning, developed and tested for years before a headline like VENOM's put it back in the news.",[327,1053,331,1055,331,1057],{"className":1054},[515,516],[339,1056,519],{"id":515},[521,1058,348,1060,348,1069,348,1080,348,1089,348,1100,348,1110,348,1120,331],{"className":1059},[524,525,526,527],[529,1061,1062,1063,593,1066],{"id":531},"\"DARPA, U.S. Air Force fly AI-controlled F-16,\" ",[534,1064,1065],{},"DARPA",[376,1067,545],{"href":1042,"target":541,"className":1068},[543,544],[529,1070,1071,1072,1075,1076],{"id":548},"A. P. Pope et al., \"Hierarchical Reinforcement Learning for Air Combat at DARPA's AlphaDogfight Trials,\" ",[534,1073,1074],{},"IEEE Transactions on Artificial Intelligence",", vol. 4, no. 6, pp. 1371-1385, 2023. DOI: ",[376,1077,545],{"href":1078,"target":541,"className":1079},"https:\u002F\u002Fdoi.org\u002F10.1109\u002FTAI.2022.3222143",[543,544],[529,1081,1082,1083,552,1085],{"id":559},"G. F. Search, \"Training Environment for High Performance Aircraft Reinforcement Learning,\" ",[534,1084,536],{},[376,1086,545],{"href":1087,"target":541,"className":1088},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2505.01953",[543,544],[529,1090,1091,1092,1095,1096],{"id":569},"R. Hasani et al., \"Liquid Time-constant Networks,\" ",[534,1093,1094],{},"Proceedings of the AAAI Conference on Artificial Intelligence",", vol. 35, no. 9, pp. 7657-7666, 2021. DOI: ",[376,1097,545],{"href":1098,"target":541,"className":1099},"https:\u002F\u002Fdoi.org\u002F10.1609\u002Faaai.v35i9.16936",[543,544],[529,1101,1102,1103,1105,1106],{"id":579},"M. Chahine et al., \"Robust flight navigation out of distribution with liquid neural networks,\" ",[534,1104,670],{},", vol. 8, no. 77, 2023. DOI: ",[376,1107,545],{"href":1108,"target":541,"className":1109},"https:\u002F\u002Fdoi.org\u002F10.1126\u002Fscirobotics.adc8892",[543,544],[529,1111,1112,1113,1115,1116],{"id":589},"\"ACE Program Achieves World First for AI in Aerospace,\" ",[534,1114,1065],{},", 2024, ",[376,1117,545],{"href":1118,"target":541,"className":1119},"https:\u002F\u002Fwww.darpa.mil\u002Fnews\u002F2024\u002Face-ai-aerospace",[543,544],[529,1121,1123,1124,593,1127],{"id":1122},"source-7","\"DARPA and US Air Force fly frontline F-16 modified for autonomous flight,\" ",[534,1125,1126],{},"FlightGlobal",[376,1128,545],{"href":1129,"target":541,"className":1130},"https:\u002F\u002Fwww.flightglobal.com\u002Farchive\u002F2026\u002F07\u002Fdarpa-and-us-air-force-fly-frontline-f-16-modified-for-autonomous-flight\u002F",[543,544],{"title":599,"searchDepth":600,"depth":600,"links":1132},[1133,1134,1135,1138,1139,1140,1141],{"id":844,"depth":600,"text":845},{"id":876,"depth":600,"text":877},{"id":916,"depth":600,"text":917,"children":1136},[1137],{"id":950,"depth":606,"text":951},{"id":969,"depth":600,"text":970},{"id":1002,"depth":600,"text":1003},{"id":1047,"depth":600,"text":1048},{"id":515,"depth":600,"text":519},"2026-08-30","DARPA's July 2026 VENOM release put AI-controlled F-16 flight back in the news. Here is a look at two distinct, peer-reviewed algorithms that this corner of autonomous flight research has actually published.",{"src":850},{"authors":1146,"badge":1149,"source":1151},[1147],{"avatar":1148,"name":622,"to":623},{"src":621},{"label":1150},"Autonomous Systems",{"name":627,"url":628},{"title":302,"description":1143},"CZSFr2Dh-26f2A3Hogk0-QPWF96SZMzwR3EFPScN-i0",1789550019217]