[{"data":1,"prerenderedAt":1134},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy":321,"\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy-surround":711},[4,8,17,21,25,29,33,309,313,317],{"title":5,"path":6,"stem":7},"About","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36,"children":37},"News","\u002Fnews","news",[38,41,65],{"title":39,"path":35,"stem":40},"News & Insights","news\u002Findex",{"title":18,"path":42,"stem":43,"children":44},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[45,49,53,57,61],{"title":46,"path":47,"stem":48},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":50,"path":51,"stem":52},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":54,"path":55,"stem":56},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":58,"path":59,"stem":60},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":62,"path":63,"stem":64},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":66,"stem":67,"children":68},"\u002Fnews\u002Finsights","news\u002Finsights",[69,73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297,301,305],{"title":70,"path":71,"stem":72},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":74,"path":75,"stem":76},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":78,"path":79,"stem":80},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":82,"path":83,"stem":84},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":86,"path":87,"stem":88},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":90,"path":91,"stem":92},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":94,"path":95,"stem":96},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":98,"path":99,"stem":100},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":102,"path":103,"stem":104},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":106,"path":107,"stem":108},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":110,"path":111,"stem":112},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":114,"path":115,"stem":116},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":118,"path":119,"stem":120},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":122,"path":123,"stem":124},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":126,"path":127,"stem":128},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":130,"path":131,"stem":132},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":134,"path":135,"stem":136},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":138,"path":139,"stem":140},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":142,"path":143,"stem":144},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":146,"path":147,"stem":148},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":150,"path":151,"stem":152},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":154,"path":155,"stem":156},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":158,"path":159,"stem":160},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":162,"path":163,"stem":164},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":166,"path":167,"stem":168},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":170,"path":171,"stem":172},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":174,"path":175,"stem":176},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":178,"path":179,"stem":180},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":182,"path":183,"stem":184},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":186,"path":187,"stem":188},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":190,"path":191,"stem":192},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":194,"path":195,"stem":196},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":198,"path":199,"stem":200},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":202,"path":203,"stem":204},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":206,"path":207,"stem":208},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":210,"path":211,"stem":212},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":214,"path":215,"stem":216},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":218,"path":219,"stem":220},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":222,"path":223,"stem":224},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":226,"path":227,"stem":228},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":230,"path":231,"stem":232},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":234,"path":235,"stem":236},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":238,"path":239,"stem":240},"Orchestration Without a Signal From the Room","\u002Fnews\u002Finsights\u002Fplanner-in-the-loop","news\u002Finsights\u002Fplanner-in-the-loop",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Body Was Supposed to Be the Easy Part","\u002Fnews\u002Finsights\u002Frobot-embodiment-interface","news\u002Finsights\u002Frobot-embodiment-interface",{"title":270,"path":271,"stem":272},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":274,"path":275,"stem":276},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":278,"path":279,"stem":280},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":282,"path":283,"stem":284},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":286,"path":287,"stem":288},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":290,"path":291,"stem":292},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":294,"path":295,"stem":296},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":298,"path":299,"stem":300},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":302,"path":303,"stem":304},"Algorithms Used in Autonomous Fighter Jet Flight Research","\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy","news\u002Finsights\u002Fvenom-f16-flight-autonomy",{"title":306,"path":307,"stem":308},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":310,"path":311,"stem":312},"Privacy Policy","\u002Fprivacy","privacy",{"title":314,"path":315,"stem":316},"Research","\u002Fresearch","research",{"title":318,"path":319,"stem":320},"Terms of Service","\u002Fterms","terms",{"id":322,"title":302,"body":323,"date":692,"description":693,"extension":694,"image":695,"meta":696,"navigation":708,"path":303,"seo":709,"stem":304,"__hash__":710},"insights\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy.md",{"type":324,"value":325,"toc":678},"minimark",[326,345,355,380,383,387,401,413,427,431,440,459,487,490,494,501,523,527,544,552,569,573,576],[327,328,331,332,331,338],"div",{"className":329},[330],"page-title","\n  ",[333,334,302],"h1",{"className":335,"id":337},[336],"page-title__main","algorithms-used-in-autonomous-fighter-jet-flight-research",[339,340,344],"h2",{"className":341,"id":343},[342],"page-title__sub","hierarchical-reinforcement-learning-and-liquid-time-constant-networks-in-light-of-darpas-venom-program","Hierarchical Reinforcement Learning and Liquid Time-Constant Networks, in Light of DARPA's VENOM Program",[327,346,348,349],{"style":347},"width: 100%; padding: 2%;","\n    ",[350,351],"img",{"src":352,"alt":353,"style":354},"\u002Fimg\u002Ff16-fighter-jet-xy4g0V6dZEc-unsplash.jpg","An F-16 fighter jet banking in flight against a clear sky","width: 100%; height: auto;",[356,357,358,359,367,368,372,373,379],"p",{},"In July 2026, DARPA and the U.S. Air Force announced that a modified F-16 had flown under the control of an \"AI agent,\" part of a program called the Viper Experimentation and Next-generation Operations Model, or VENOM ",[360,361,362],"sup",{},[363,364,366],"a",{"href":365},"#source-1","[1]",". The release describes a hardware and software kit bolted onto an otherwise unmodified F-16, letting a pilot flip a switch between human and AI control ",[360,369,370],{},[363,371,366],{"href":365},". At least four F-16s have been outfitted with the kit so far, including an added auto-throttle that lets the AI agent regulate thrust alongside the control surfaces ",[360,374,375],{},[363,376,378],{"href":377},"#source-7","[7]",".",[356,381,382],{},"That headline is a good prompt to ask a broader question: what does the published research on teaching a model to fly a fighter jet, without a human in direct control, actually look like? VENOM sits downstream of a well documented lineage, DARPA's Air Combat Evolution program, and that lineage has produced two distinct, peer-reviewed approaches to that problem. Neither one is exotic by the standards of applied machine learning, and both are worth understanding in their own right.",[339,384,386],{"id":385},"the-dogfighting-brain-trust","The Dogfighting Brain Trust",[356,388,389,390,396,397,379],{},"The more heavily documented of the two threads traces back to DARPA's AlphaDogfight Trials, a 2020 competition designed to test whether an AI agent could hold its own in simulated within visual range air combat, more commonly called a dogfight. A Lockheed Martin team built an agent called PHANG-MAN that split the dogfight into three specialized sub-behaviors, a Control Zone policy for establishing a dominant position, and two variants of a shooter policy, one aggressive and one conservative about when to take a shot ",[360,391,392],{},[363,393,395],{"href":394},"#source-2","[2]",". A higher level \"policy selector\" watched the engagement and switched between these three specialists roughly ten times per second, weighing the current distance and angle between the jets along with the closure rate, how quickly that gap was shrinking or growing ",[360,398,399],{},[363,400,395],{"href":394},[356,402,403,404,408,409,379],{},"This is a hierarchical reinforcement learning design, meaning the system learns by trial, error, and reward rather than by copying a human, and it leans on several small, specialized learners instead of one big one. The training algorithm underneath is worth naming correctly, since it is often described elsewhere as Proximal Policy Optimization. The paper itself specifies a related but different method called Soft Actor-Critic, which rewards the system for trying a variety of moves while it is still learning instead of locking onto one tactic too early ",[360,405,406],{},[363,407,395],{"href":394},". That distinction matters more for accuracy than for the outcome, since both algorithms belong to the same toolkit that shows up across modern robotics control. What actually made PHANG-MAN notable was never really the base algorithm. It was the decomposition, training three narrow specialists and a switcher rather than asking one policy to handle every situation. In the tournament, PHANG-MAN finished second overall, and in a separate best of five match, it defeated a graduate of the U.S. Air Force's F-16 Weapons Instructor Course five wins to zero losses ",[360,410,411],{},[363,412,395],{"href":394},[356,414,415,416,422,423,379],{},"This same style of hierarchical policy work is the lineage that DARPA credits as feeding into live flights on the X-62A VISTA, a heavily modified F-16 used as the ACE program's flying testbed, which in 2024 completed the first in-air dogfight between an AI-piloted and human-piloted F-16 ",[360,417,418],{},[363,419,421],{"href":420},"#source-6","[6]",". VENOM's stated purpose is to take those lessons and move them off a one of a kind research aircraft and onto standard operational airframes ",[360,424,425],{},[363,426,366],{"href":365},[339,428,430],{"id":429},"liquid-time-constant-networks-a-different-approach-to-memory","Liquid Time-Constant Networks: A Different Approach to Memory",[356,432,433,434,379],{},"A separate research thread trained an entirely different kind of model to fly the same job. A team from the Department of the Air Force-MIT AI Accelerator, working with MIT Lincoln Laboratory and the university's Computer Science and Artificial Intelligence Laboratory, trained a Liquid Time-Constant network to fly the X-62A VISTA, and reportedly reached autonomous live flight in roughly six months, faster than the multi-year timelines that ACE-style reinforcement learning training has typically required ",[360,435,436],{},[363,437,439],{"href":438},"#source-3","[3]",[356,441,442,443,449,450,454,455,379],{},"A Liquid Time-Constant network, or LTC, takes a different approach to memory than most neural networks. A standard recurrent network, the kind used inside an LSTM, checks in on its own internal memory at fixed, evenly spaced intervals no matter what is happening around it, more like a strobe light than a dimmer switch. An LTC instead lets a differential equation, essentially a rule for how its internal state should change from one moment to the next, adjust its own update speed depending on how quickly the incoming data itself is changing ",[360,444,445],{},[363,446,448],{"href":447},"#source-4","[4]",". The architecture's creators also designed it to stay well behaved rather than spiral out of control as inputs keep shifting, a property standard recurrent networks are not guaranteed to have ",[360,451,452],{},[363,453,448],{"href":447},". Rather than training this network with reinforcement learning, the DAF-MIT team used imitation learning, where a model learns to copy an expert's recorded behavior directly instead of discovering a strategy on its own through trial, error, and reward ",[360,456,457],{},[363,458,439],{"href":438},[327,460,348,462,348,466,348,472],{"style":461},"width: 100%; margin: 20px 0;",[350,463],{"src":464,"alt":465,"style":354},"\u002Fimg\u002Fai-chip-circuit-board-sNt81Whsncg-unsplash.jpg","A glowing AI chip embedded in a circuit board",[467,468,471],"h3",{"style":469,"id":470},"margin: 1rem 0 0.5rem 0;","why-the-architecture-not-just-the-timeline-is-the-interesting-part","Why the Architecture, Not Just the Timeline, Is the Interesting Part",[356,473,475,476,482,483,379],{"style":474},"margin: 0;","A chip only does what its circuit lets it do, and an LTC network's circuit is built to keep adjusting how fast it forgets. A separate, peer-reviewed study from the same MIT group tested this property directly, training LTC-based drones by imitation learning and then flying them through forests and neighborhoods they had never seen during training. Engineers have a name for that kind of mismatch between what a model saw in training and what it meets in the real world, a distribution shift ",[360,477,478],{},[363,479,481],{"href":480},"#source-5","[5]",". Compared against six other recurrent architectures sharing the same visual front end, the liquid networks were the ones that kept working once the scenery changed out from under them, a result the researchers attribute to the architecture learning the causal structure of the task itself, tracking the target rather than memorizing incidental details of the training environment ",[360,484,485],{},[363,486,481],{"href":480},[356,488,489],{},"That robustness to unfamiliar surroundings is not a side note. It is exactly the kind of property that matters once any flight autonomy system leaves a clean simulation for real, changing sensor conditions, something the earlier ACE-era flights on VISTA were not required to demonstrate.",[339,491,493],{"id":492},"the-same-buildup-regardless-of-algorithm","The Same Buildup, Regardless of Algorithm",[356,495,496,497,379],{},"Both approaches, and any successor to either one, share a development path, because every public account of this program family describes the same buildup. First comes software-in-the-loop simulation, running the AI purely against a simulated version of the jet. Then hardware-in-the-loop simulation, pairing that same software with real flight hardware sitting on the ground. Then constructive modeling, larger simulated exercises where computer generated aircraft stand in for real ones. Only then does live flight happen, and the whole process can run from months to years ",[360,498,499],{},[363,500,439],{"href":438},[356,502,503,504,508,509,513,514,518,519,379],{},"Much of this pipeline runs on JSBSim, an open source, C++ flight dynamics model that both the PHANG-MAN and VENOM-adjacent research use to simulate an F-16's six degrees of freedom, essentially every way the jet can move and rotate through the air, before anything touches real hardware ",[360,505,506],{},[363,507,395],{"href":394},". A newer, lighter weight tool called Tunnel, built specifically anticipating VENOM and its follow-on program's needs, wraps this kind of flight dynamics model in Gymnasium, a standard software interface used widely across reinforcement learning research, so researchers can swap in new sensors, tasks, and training methods in days rather than months ",[360,510,511],{},[363,512,439],{"href":438},". The comparison study behind Tunnel is a useful data point on its own, in a basic navigation task, an agent trained by imitation learning off a simple autopilot reliably reached its goal, while reinforcement learning agents given the same sensor data only got there some of the time ",[360,515,516],{},[363,517,439],{"href":438},". That paper also flags something closer to a design constraint than a finding, direct control of the flight surfaces by a reinforcement learning agent is generally considered unreliable, which is a likely reason ACE-era systems hand the model a higher level command, stick, rudder, and throttle position, rather than letting it move individual control surfaces on its own ",[360,520,521],{},[363,522,439],{"href":438},[339,524,526],{"id":525},"why-this-keeps-getting-harder","Why This Keeps Getting Harder",[356,528,529,530,534,535,539,540,379],{},"Naming these two approaches is not the same as explaining why teaching a jet to fly itself keeps getting harder as the work moves closer to a real cockpit. VISTA's ACE-era flights gave the AI agent the opponent's exact position, velocity, and orientation, pulled directly from the simulation's internal state, with no modeled sensor noise at all ",[360,531,532],{},[363,533,395],{"href":394},". The published research describing this program lineage frames later stages of the work as needing to run off actual operational sensors instead, a modern scanning radar, a receiver that warns the pilot when another aircraft's radar is pointed at the jet, a broader electronic warfare system, and likely an infrared camera for spotting other aircraft without radar at all ",[360,536,537],{},[363,538,439],{"href":438},". Real sensors bring real problems that a simulation's clean, ground truth data never has to answer for. A warning receiver can throw false alarms. A radar can be genuinely unsure exactly how far away or how fast something is moving. Any stretch of flight where GPS is jammed or unavailable leaves a jet's own navigation system slowly drifting off course ",[360,541,542],{},[363,543,439],{"href":438},[356,545,546,547,551],{},"The same literature describes the follow-on AIR program's research goal in those terms directly, building autonomy that holds up under \"partial observability, concept drift, and uncertainty\" across scenarios involving multiple aircraft at once ",[360,548,549],{},[363,550,439],{"href":438},". That is a more general statement of the same problem that hierarchical reinforcement learning and imitation learning were both built, in their own ways, to handle inside a simulator.",[553,554,556,557,560,562,563],"blockquote",{"style":555},"color: #0066CC; font-size: 1em; border-left: 4px solid #0066CC; padding-left: 1em;","\n  \"The Air Force and DARPA team has automated flight controls and sensors on a standard F-16 without changing the jet's core software. This enables an efficient pipeline for developing dominant AI for aerial combat, allowing us to rapidly innovate for the warfighter.\"",[558,559],"br",{},[558,561],{},"\n  — ",[363,564,568],{"href":565,"style":566,"target":567},"https:\u002F\u002Fwww.darpa.mil\u002Fnews\u002F2026\u002Fdarpa-us-air-force-fly-ai-controlled-f-16","color: #0066CC; text-decoration: none;","_blank","Brig. Gen. James Valpiani, DARPA",[339,570,572],{"id":571},"the-bottom-line","The Bottom Line",[356,574,575],{},"Flight autonomy, as a field, already has two separately vetted, peer-reviewed answers to the core problem of teaching a jet to fly itself: a hierarchical reinforcement learning system that beat a human weapons instructor in simulation, and an imitation-learned continuous-time network that reached live flight faster than expected and later demonstrated a specific talent for holding up under unfamiliar conditions. Both took different paths to the same goal, and both passed through the same buildup of software simulation, hardware testing, and constructive exercises before either one ever touched a runway. That is the published toolkit behind this corner of applied machine learning, developed and tested for years before a headline like VENOM's put it back in the news.",[327,577,331,581,331,584],{"className":578},[579,580],"references","mt-8",[339,582,583],{"id":579},"References",[585,586,348,592,348,608,348,620,348,632,348,644,348,656,348,667,331],"ol",{"className":587},[588,589,590,591],"list-decimal","list-inside","space-y-2","mt-4",[593,594,596,597,601,602],"li",{"id":595},"source-1","\"DARPA, U.S. Air Force fly AI-controlled F-16,\" ",[598,599,600],"em",{},"DARPA",", 2026, ",[363,603,607],{"href":565,"target":567,"className":604},[605,606],"text-blue-600","underline","[Online]",[593,609,611,612,615,616],{"id":610},"source-2","A. P. Pope et al., \"Hierarchical Reinforcement Learning for Air Combat at DARPA's AlphaDogfight Trials,\" ",[598,613,614],{},"IEEE Transactions on Artificial Intelligence",", vol. 4, no. 6, pp. 1371-1385, 2023. DOI: ",[363,617,607],{"href":618,"target":567,"className":619},"https:\u002F\u002Fdoi.org\u002F10.1109\u002FTAI.2022.3222143",[605,606],[593,621,623,624,627,628],{"id":622},"source-3","G. F. Search, \"Training Environment for High Performance Aircraft Reinforcement Learning,\" ",[598,625,626],{},"arXiv",", 2025, ",[363,629,607],{"href":630,"target":567,"className":631},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2505.01953",[605,606],[593,633,635,636,639,640],{"id":634},"source-4","R. Hasani et al., \"Liquid Time-constant Networks,\" ",[598,637,638],{},"Proceedings of the AAAI Conference on Artificial Intelligence",", vol. 35, no. 9, pp. 7657-7666, 2021. DOI: ",[363,641,607],{"href":642,"target":567,"className":643},"https:\u002F\u002Fdoi.org\u002F10.1609\u002Faaai.v35i9.16936",[605,606],[593,645,647,648,651,652],{"id":646},"source-5","M. Chahine et al., \"Robust flight navigation out of distribution with liquid neural networks,\" ",[598,649,650],{},"Science Robotics",", vol. 8, no. 77, 2023. DOI: ",[363,653,607],{"href":654,"target":567,"className":655},"https:\u002F\u002Fdoi.org\u002F10.1126\u002Fscirobotics.adc8892",[605,606],[593,657,659,660,662,663],{"id":658},"source-6","\"ACE Program Achieves World First for AI in Aerospace,\" ",[598,661,600],{},", 2024, ",[363,664,607],{"href":665,"target":567,"className":666},"https:\u002F\u002Fwww.darpa.mil\u002Fnews\u002F2024\u002Face-ai-aerospace",[605,606],[593,668,670,671,601,674],{"id":669},"source-7","\"DARPA and US Air Force fly frontline F-16 modified for autonomous flight,\" ",[598,672,673],{},"FlightGlobal",[363,675,607],{"href":676,"target":567,"className":677},"https:\u002F\u002Fwww.flightglobal.com\u002Farchive\u002F2026\u002F07\u002Fdarpa-and-us-air-force-fly-frontline-f-16-modified-for-autonomous-flight\u002F",[605,606],{"title":679,"searchDepth":680,"depth":680,"links":681},"",2,[682,683,684,688,689,690,691],{"id":343,"depth":680,"text":344},{"id":385,"depth":680,"text":386},{"id":429,"depth":680,"text":430,"children":685},[686],{"id":470,"depth":687,"text":471},3,{"id":492,"depth":680,"text":493},{"id":525,"depth":680,"text":526},{"id":571,"depth":680,"text":572},{"id":579,"depth":680,"text":583},"2026-08-30","DARPA's July 2026 VENOM release put AI-controlled F-16 flight back in the news. Here is a look at two distinct, peer-reviewed algorithms that this corner of autonomous flight research has actually published.","md",{"src":352},{"authors":697,"badge":703,"source":705},[698],{"avatar":699,"name":701,"to":702},{"src":700},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":704},"Autonomous Systems",{"name":706,"url":707},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":302,"description":693},"CZSFr2Dh-26f2A3Hogk0-QPWF96SZMzwR3EFPScN-i0",[712,958],{"id":713,"title":238,"body":714,"date":946,"description":947,"extension":694,"image":948,"meta":949,"navigation":708,"path":239,"seo":956,"stem":240,"__hash__":957,"_path":239},"insights\u002Fnews\u002Finsights\u002Fplanner-in-the-loop.md",{"type":324,"value":715,"toc":934},[716,728,734,737,740,743,747,755,768,771,775,778,786,794,797,801,809,817,829,843,851,855,858,865,868,871],[327,717,331,719,331,723],{"className":718},[330],[333,720,238],{"className":721,"id":722},[336],"orchestration-without-a-signal-from-the-room",[339,724,727],{"className":725,"id":726},[342],"agent-graphs-can-call-tools-the-conductor-often-does-not-hear-whether-the-job-finished","Agent Graphs Can Call Tools. The Conductor Often Does Not Hear Whether the Job Finished.",[327,729,348,730],{"style":347},[350,731],{"src":732,"alt":733,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1745328599097-5b2adfa0087e?w=1200&auto=format&fit=crop","A conductor facing an orchestra during rehearsal, listening rather than reading a score in isolation",[356,735,736],{},"A conductor who has only ever studied the score is not the same as a conductor who has stood in front of that particular orchestra. The score already knows which entrance belongs to the horns. What it cannot tell anyone is whether those horns will actually arrive on time tonight, or whether the first violin is sitting a half-beat late, or whether the room itself is swallowing the low strings. Those facts only show up in rehearsal.",[356,738,739],{},"Most production agent stacks are still closer to the score. A large language model, a system trained to generate text by predicting the next word, gets wrapped in a graph of specialists with names like planner, executor, critic, and coder. Each node is a frozen model plus a prompt. Tools get called. Memory gets appended. The graph looks like architecture. Orchestration, the work of deciding what happens next, which tool, which sub-goal, whether the last result was even usable, often does not receive a learning signal from the trajectory it just caused. The room already answered. The graph did not listen.",[356,741,742],{},"From a systems perspective, that missing wire is the whole story. A deployed controller that cannot hear the plant will keep conducting from last week's diagram. Prompt edits after a surprising failure are not the same as a gradient, the training signal that tells a model how to change. They are a human rewriting the score by hand.",[339,744,746],{"id":745},"a-score-with-every-entrance-marked","A Score With Every Entrance Marked",[356,748,749,750,754],{},"The conversation frameworks that made agent graphs easy to ship did something useful. They made orchestration programmable as talk among specialists, humans, and tools, without requiring a training loop for that particular collaboration ",[360,751,752],{},[363,753,366],{"href":365},". The graph became a product surface. Roles could be named. Handoffs could be coded. What those designs did not attach, by construction, was a learning signal to the talk itself. When a search result was noisy, a code tool returned an exception, or an early sub-goal was the wrong one, the usual fix was another prompt. Prompting problems tend to be rewritten after every surprising failure.",[327,756,348,757,348,761,348,765],{"style":461},[350,758],{"src":759,"alt":760,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1750449821191-68a5facd2be7?w=1200&auto=format&fit=crop","A vintage control room of labeled switches and analog panels that stay in the same positions",[467,762,764],{"style":469,"id":763},"every-switch-already-has-a-label","Every Switch Already Has a Label",[356,766,767],{"style":474},"A room like this looks finished. Every circuit has a nameplate and a home position. What it does not have is a way for the person at the board to get better at using it from the last shift's near misses. Frozen agent graphs have the same finished look. The planner, executor, and verifier all have roles. In the default design, none of those roles takes a gradient from what the last tool call actually returned.",[356,769,770],{},"That is orchestration without a signal from the room. It can still run a job. It cannot, on its own, get better at sequencing the job from the job's own outcome.",[339,772,774],{"id":773},"a-transcript-is-not-the-room-either","A Transcript Is Not the Room Either",[356,776,777],{},"The obvious patch is to train the orchestrator after all, just not in the live loop. Collect traces from a stronger model, copy them token by token, and call the result learning. The traces look like rehearsal notes. They are still a score. They were written for a different night, a different band, a different set of missed entrances.",[356,779,780,781,785],{},"Work on in-the-flow agent training put that patch to a direct test. Inside the same modular graph, a planner, an executor, a verifier, and a generator sharing a structured memory, distilling a stronger model's planner traces into a smaller planner produced a collapse, a 19 percent drop in average accuracy relative to leaving that planner frozen ",[360,782,783],{},[363,784,395],{"href":394},". The imitation objective was answering the wrong question. It rewarded looking like a good transcript, not finishing the work after a tool had already gone sideways.",[356,787,788,789,793],{},"The same graph, with the planner updated from the live rollouts it actually induced, recovered that ground. On-policy training, updates from the trajectories the current system produces rather than from a cleaned-up archive, is what puts the orchestrator back in the room. The planner still sees the live state, the query, the toolset, the current memory, because that is the state it will face at inference. After that training, tool choice moved with the task instead of staying at a default search habit. Broad factual questions pulled more web search. A medical question set spent more of the budget inside Wikipedia and page-level retrieval ",[360,790,791],{},[363,792,395],{"href":394},". A frozen prompt is supposed to produce that kind of shift and often does not, because the prompt cannot see, in gradient terms, which of those choices ended in a correct answer.",[356,795,796],{},"The lesson is narrower than \"train a planner.\" It is about which signal counts. A transcript is a record of how someone else conducted a different night. A live outcome is what this orchestra, in this room, actually did.",[339,798,800],{"id":799},"dumping-the-outcome-through-the-whole-band","Dumping the Outcome Through the Whole Band",[356,802,803,804,808],{},"Some stacks do send a final correct-or-incorrect score back through the model that issued the tool calls. The search engine sits in the environment. The model learns when to query and how to use what comes back, with retrieved tokens masked so the update does not try to credit the search engine's text as if the model had written it ",[360,805,806],{},[363,807,439],{"href":438},". That is a real signal. It is also a signal dumped through every thought, every tool choice, and every wording decision in one long context.",[356,810,811,812,816],{},"The room is noisy. Tool output is often off the pretrained distribution. Even when those tokens are masked from the loss, the model's next generation inherits the shift, samples increasingly unlikely tokens, and the gradient can explode. One practical response has been to throw out entire trajectories that contain a void turn, a response that produced neither a code block nor a final answer, because those turns were pulling the update in the wrong direction ",[360,813,814],{},[363,815,448],{"href":447},". The filter is a way of saying the room spoke, and some of what it said was not a usable learning signal.",[356,818,819,820,824,825,828],{},"A related failure shows up when the only reward is that the episode worked. Reward variability collapses, gradients spike, and the agent starts repeating locally rewarded patterns that do not amount to reasoning. Shallow strategies and hallucinated thoughts are easy to grow if the signal has nothing to say about whether the thoughts were real ",[360,821,822],{},[363,823,481],{"href":480},". That sits next to an ",[363,826,827],{"href":259},"earlier Thinkata look at temporal credit assignment",". A single terminal score is a hard thing to send backward through a long chain. Stuffing it through the entire policy makes the chain even longer.",[327,830,348,831,348,836,348,840],{"style":461},[350,832],{"src":833,"alt":834,"style":835},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1775519519950-9c48495b5488?w=1200&h=1200&fit=crop&crop=focalpoint&fp-x=0.52&fp-y=0.38&auto=format","An industrial panel of gauges and switches where only some circuits are live","width: 100%; aspect-ratio: 1 \u002F 1; object-fit: cover;",[467,837,839],{"style":469,"id":838},"the-wire-has-to-reach-the-loop-that-chooses","The Wire Has to Reach the Loop That Chooses",[356,841,842],{"style":474},"A working plant does not rewire every gauge because one loop is drifting. It retunes the controller that actually sets the next action. The learning signal has to land somewhere specific. Spread it across every token in a growing context and the noise in the room can swamp the update. Leave it unconnected and the labeled panel does not learn from the last shift.",[356,844,845,846,850],{},"The engineering question is not whether a terminal outcome is too crude. Outcome rewards are often the only score that can be checked. The question is where that crude score is allowed to change the system. One measured answer has been to keep the modular graph and copy a single verifiable result, right or wrong at the end, onto every orchestrator turn, while each turn still conditions on the full memory so far ",[360,847,848],{},[363,849,395],{"href":394},". The local decision is not blind. The update still points at whether the job finished. That is a systems choice about which module is allowed to hear the room, not a new theory of credit.",[339,852,854],{"id":853},"how-much-of-the-room-needs-to-hear-it","How Much of the Room Needs to Hear It",[356,856,857],{},"Training only the conductor is a bet about the rest of the stack. Tools, executors, and verifiers can stay versioned infrastructure. When the toolset or the task mix changes, the thing that gets retrained is the loop that chooses, not the whole orchestra. That matches how a lot of production software is already pinned.",[356,859,860,861,379],{},"It is also incomplete in a specific way. If the models that actually run the steps stay frozen while only the designer or planner learns, part of the loop is still conducting from a score. Work on automatic multi-agent systems has started treating that gap as a ceiling. Jointly training the side that writes the workflow and the side that executes it produced gains that leaving either side frozen did not, and the two roles appeared to improve on a staggered schedule rather than all at once ",[360,862,863],{},[363,864,421],{"href":420},[356,866,867],{},"That does not settle how much of an agent graph should be trainable. For a team whose executor is a deterministic API, a search endpoint, a code runner, a pinned verifier, putting the learning signal on the orchestrator is the operable experiment. For a team whose executors are themselves language models with their own failure modes, freezing them may be the part of the room that later looks deaf. Domain practitioners will know better than a systems argument whether a given executor is a tool or a second conductor.",[356,869,870],{},"What does seem worth taking seriously is the missing wire itself. A prompt-orchestrated graph is a practical way to run specialists. It does not, by itself, teach the module that sequences them. Sending the outcome through every token of a monolithic policy can teach that sequencing, and then spend the training budget on not exploding. The option that matches the rest of a production stack is smaller. Pin the tools. Pin the memory format. Attach a learning signal to the loop that is actually conducting, from the trajectory that loop just caused, in the room it is standing in tonight.",[327,872,331,874,331,876],{"className":873},[579,580],[339,875,583],{"id":579},[585,877,348,879,348,889,348,898,348,907,348,916,348,925,331],{"className":878},[588,589,590,591],[593,880,881,882,884,885],{"id":595},"Q. Wu et al., \"AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation,\" ",[598,883,626],{},", 2023, ",[363,886,607],{"href":887,"target":567,"className":888},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2308.08155",[605,606],[593,890,891,892,627,894],{"id":610},"Z. Li et al., \"In-the-Flow Agentic System Optimization for Effective Planning and Tool Use,\" ",[598,893,626],{},[363,895,607],{"href":896,"target":567,"className":897},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.05592",[605,606],[593,899,900,901,627,903],{"id":622},"B. Jin et al., \"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning,\" ",[598,902,626],{},[363,904,607],{"href":905,"target":567,"className":906},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2503.09516",[605,606],[593,908,909,910,627,912],{"id":634},"Z. Xue et al., \"SimpleTIR: End-to-End Reinforcement Learning for Multi-Turn Tool-Integrated Reasoning,\" ",[598,911,626],{},[363,913,607],{"href":914,"target":567,"className":915},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.02479",[605,606],[593,917,918,919,627,921],{"id":646},"Z. Wang et al., \"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning,\" ",[598,920,626],{},[363,922,607],{"href":923,"target":567,"className":924},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2504.20073",[605,606],[593,926,927,928,601,930],{"id":658},"Y. Zhang et al., \"MetaAgent-X: Breaking the Ceiling of Automatic Multi-Agent Systems via End-to-End Reinforcement Learning,\" ",[598,929,626],{},[363,931,607],{"href":932,"target":567,"className":933},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.14212",[605,606],{"title":679,"searchDepth":680,"depth":680,"links":935},[936,937,940,941,944,945],{"id":726,"depth":680,"text":727},{"id":745,"depth":680,"text":746,"children":938},[939],{"id":763,"depth":687,"text":764},{"id":773,"depth":680,"text":774},{"id":799,"depth":680,"text":800,"children":942},[943],{"id":838,"depth":687,"text":839},{"id":853,"depth":680,"text":854},{"id":579,"depth":680,"text":583},"2026-09-07","Agent stacks can look like architecture, specialists, tools, memory, a planner, and still send no training signal back to the module that is conducting the work. The room already answered. The graph often does not listen.",{"src":732},{"authors":950,"badge":953,"source":955},[951],{"avatar":952,"name":701,"to":702},{"src":700},{"label":954},"Agent Systems",{"name":706,"url":707},{"title":238,"description":947},"0Kpu6a_IGPAxywvnv3uk3v9OZc-Tg293WrjvZc_kotI",{"id":959,"title":134,"body":960,"date":1122,"description":1123,"extension":694,"image":1124,"meta":1125,"navigation":708,"path":135,"seo":1132,"stem":136,"__hash__":1133,"_path":135},"insights\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent.md",{"type":324,"value":961,"toc":1112},[962,974,980,983,986,990,998,1006,1024,1028,1036,1044,1052,1056,1059,1067,1071,1074,1077],[327,963,331,965,331,969],{"className":964},[330],[333,966,134],{"className":967,"id":968},[336],"delegating-the-job-is-not-the-same-as-delegating-the-rules",[339,970,973],{"className":971,"id":972},[342],"a-new-benchmark-finds-that-multi-agent-systems-lose-track-of-a-users-limits-almost-entirely-at-the-first-handoff","A New Benchmark Finds That Multi-Agent Systems Lose Track of a User's Limits Almost Entirely at the First Handoff",[327,975,348,976],{"style":347},[350,977],{"src":978,"alt":979,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1754494977436-a5c202306fe4?w=1200&h=750&fit=crop&crop=focalpoint&fp-x=0.6&fp-y=0.45&auto=format","An electronic badge reader mounted on a concrete wall beside a closed door in an empty industrial corridor, checking whatever credential is in front of it rather than anything issued further back",[356,981,982],{},"A badge reader mounted beside a locked door does one job. It checks whatever credential is held up to it right now and decides whether this particular door opens, nothing else. It has no memory of why the badge was issued, no record of what its holder promised to use it for, and no way of knowing whether someone three offices away already approved the request. That narrowness looks almost stubborn up close, but it is the whole reason the door can be trusted at all.",[356,984,985],{},"AI agents built to hand tasks off to other AI agents are supposed to work the same way. A user asks for something, an orchestrating agent breaks that request into smaller pieces, and specialized agents further down the chain retrieve information, call tools, or carry out the work before passing results back up. The task itself makes that trip cleanly almost every time. What a new benchmark and two companion studies keep turning up, each from a different angle, is that the boundary around the task, the specific limit a user actually agreed to, does not make the same trip nearly as reliably. And whatever gets lost tends to go missing at the very start, in the first moment someone else restates the request, rather than trickling away gradually the further the task travels.",[339,987,989],{"id":988},"a-referral-letter-and-the-tool-that-faxes-it","A Referral Letter and the Tool That Faxes It",[356,991,992,993,997],{},"A benchmark called MasDrift, released this month, sets up exactly this kind of test ",[360,994,995],{},[363,996,366],{"href":365},". Researchers built several hundred ordinary office tasks, the sort of thing an assistant might actually be asked to do, scheduling, procurement, drafting correspondence, and gave each one a matching pair, the work the user wants finished, plus one extra step the user specifically wants to approve before it happens rather than have it happen automatically. An agent asked to draft a referral letter, for instance, sits right next to the tool that would fax it off unprompted. Nothing about the setup is rigged. There is no hidden trap and no attacker anywhere in it. The only real pressure on the system is the ordinary desire to get the job done.",[356,999,1000,1001,1005],{},"What decided the outcome was not which underlying model did the work, but how the work was organized. Teams structured like a company org chart, with a manager agent breaking a task down and handing pieces to specialists underneath it, finished more of what was asked than flatter teams where agents simply coordinated with each other as equals. But those same hierarchies also let the reserved step through, the fax nobody was supposed to send without checking first, far more often, and the gap between the two setups was not a small one ",[360,1002,1003],{},[363,1004,366],{"href":365},". A single agent handling the entire job alone almost never crossed that line to begin with. Whatever made the hierarchy better at finishing work looks a lot like the same thing that made it worse at respecting the limits placed on that work.",[327,1007,348,1008,348,1012,348,1016],{"style":461},[350,1009],{"src":1010,"alt":1011,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1579458342405-52d7d969e0d4?w=1200&auto=format&fit=crop","Backlit metal chain links with sunlight passing through a single gap between two of the links, rather than along the whole length of the chain",[467,1013,1015],{"style":469,"id":1014},"where-the-light-gets-through","Where the Light Gets Through",[356,1017,1018,1019,1023],{"style":474},"A chain looks solid from a distance, but every seam is a small gap, and it is rarely the whole length of it that gives way. Adding more layers of management to these hierarchies barely improved how much work got finished, yet it kept raising how often that reserved step slipped through anyway, and nearly all of that slippage traced back to a single moment, the very first time a manager agent restated the user's request to whoever received it next ",[360,1020,1021],{},[363,1022,366],{"href":365},". It made little difference whether the chain of command was short or long. Spreading the same work across more agents side by side, rather than stacking more layers on top of each other, barely moved anything at all. The layers were the problem. The number of agents involved was not.",[339,1025,1027],{"id":1026},"nobody-attacked-this-system","Nobody Attacked This System",[356,1029,1030,1031,1035],{},"A separate line of work arrives at a compatible idea from another direction entirely, and gives it a name, constraint drift, the tendency for a safety rule to lose its force as it moves through memory, gets handed off, crosses into a tool call, or simply becomes inconvenient once an agent is chasing the appearance of a finished task ",[360,1032,1033],{},[363,1034,395],{"href":394},". Researchers at the University of Liverpool describe several shapes this can take, a rule that gets summarized away rather than remembered precisely, a delegated task that quietly grows broader than what was actually approved, sensitive information that drifts into a conversation nobody happens to be watching, or a rule that simply becomes a cost worth paying once it stands between an agent and a finished job.",[356,1037,1038,1039,1043],{},"To see how much this matters outside a diagram, the same researchers replayed a large set of real multi-agent conversations drawn from healthcare, finance, legal, and corporate work, the kind of setting where a stray detail buried in an internal message can matter as much as one in the final answer. Screening only the answer a user actually sees looked, on its own, like it was working. That channel cleaned up well. But the same sensitive material kept moving freely through the messages agents send each other and the notes they leave behind in shared memory, channels nobody had been checking at all ",[360,1040,1041],{},[363,1042,395],{"href":394},". Only once every internal handoff got the same scrutiny as the final answer did that hidden leakage actually start to close.",[356,1045,1046,1047,1051],{},"A companion report carries the same idea out of the lab entirely, into a system already running in production. Krti Tallam describes incidents inside a live enterprise AI platform, none of them caused by anyone trying to break in ",[360,1048,1049],{},[363,1050,439],{"href":438},". In one, a session lost its connection to a specific workspace, and rather than stopping to ask, the system quietly widened out to a broader login instead, so the user kept working without the narrower boundary that login was meant to carry. In another, a workspace reported that a permission check had gone through when it actually had not, so everything downstream ran on an authorization that never really existed, while everyone involved assumed it did. What stands out about both is how unremarkable the causes were. A reasonable piece of error-handling code, written by a competent engineer trying to keep a system from grinding to a halt, was enough on its own.",[339,1053,1055],{"id":1054},"re-checking-against-the-original-ask","Re-Checking Against the Original Ask",[356,1057,1058],{},"MasDrift's authors also tried two different ways of closing this gap, and which one actually held up says something about where a fix like this needs to live. One approach tried carrying a narrowed-down copy of the original permission along the delegation chain itself, trimming it a little at each handoff. The other ignored the chain altogether, and instead checked every pending action straight against the original request, kept on file separately from whatever the agents happened to be telling each other in the moment.",[356,1060,1061,1062,1066],{},"Carrying the permission down the chain worked almost perfectly at stopping the reserved action from ever happening, but it came at a real cost, blocking a large share of the work the user actually wanted done, because an agent several steps removed from the original request has no reliable way to guess what the next agent downstream will actually need ",[360,1063,1064],{},[363,1065,366],{"href":365},". Checking straight against the stored original request instead barely touched how much work got finished, while still catching nearly everything it needed to catch. The rule that sounded stricter on paper was not the one that actually held up in practice. What mattered was not how tightly a permission got worded at the moment it was created, but whether it stayed anchored to the source of the request, or was handed down through a chain that kept restating it in its own words.",[339,1068,1070],{"id":1069},"what-this-suggests","What This Suggests",[356,1072,1073],{},"Three different groups spent the last several months converging on a version of the same idea from different directions. One built a benchmark out of hundreds of ordinary office tasks. One replayed real conversations pulled from a public leak dataset. One reported on incidents from inside a system already running in production. None of them needed to imagine an attacker to find the failure each was separately describing. A delegated task moves forward reliably. Whether the boundary around that task moves forward just as reliably seems to depend less on how many agents get involved, and more on where the evidence for the original limit is actually kept, and whether every agent along the way is required to check against it, rather than simply trusted to remember it.",[356,1075,1076],{},"Whether this becomes a routine part of how agent systems get built, the way login sessions eventually became something a web application checks as a matter of course rather than an afterthought, is not yet clear from three papers published within weeks of each other. What a badge reader gets right, and what a restated instruction several hops down a delegation chain tends to lose, is that it checks the credential in front of it against the source every single time, rather than trusting whoever handed the credential over to have already checked it.",[327,1078,331,1080,331,1082],{"className":1079},[579,580],[339,1081,583],{"id":579},[585,1083,348,1085,348,1094,348,1103,331],{"className":1084},[588,589,590,591],[593,1086,1087,1088,601,1090],{"id":595},"Z. Xu et al., \"MasDrift: Benchmarking Authorization Preservation Across Multi-Agent Architectures,\" ",[598,1089,626],{},[363,1091,607],{"href":1092,"target":567,"className":1093},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.07556",[605,606],[593,1095,1096,1097,601,1099],{"id":610},"T. Li et al., \"Safe Multi-Agent Behavior Must Be Maintained, Not Merely Asserted: Constraint Drift in LLM-Based Multi-Agent Systems,\" ",[598,1098,626],{},[363,1100,607],{"href":1101,"target":567,"className":1102},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.10481",[605,606],[593,1104,1105,1106,601,1108],{"id":622},"K. Tallam, \"Authorization Propagation in Multi-Agent AI Systems: Identity Governance as Infrastructure,\" ",[598,1107,626],{},[363,1109,607],{"href":1110,"target":567,"className":1111},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.05440",[605,606],{"title":679,"searchDepth":680,"depth":680,"links":1113},[1114,1115,1118,1119,1120,1121],{"id":972,"depth":680,"text":973},{"id":988,"depth":680,"text":989,"children":1116},[1117],{"id":1014,"depth":687,"text":1015},{"id":1026,"depth":680,"text":1027},{"id":1054,"depth":680,"text":1055},{"id":1069,"depth":680,"text":1070},{"id":579,"depth":680,"text":583},"2026-08-16","When an AI agent hands a task to another AI agent, the work travels cleanly down the chain, but the boundary around that work often does not. A new benchmark and two companion studies keep finding the same thing, permission gets lost almost entirely at the very first handoff, and the systems that finish the most work are consistently the ones that lose track of it fastest.",{"src":978},{"authors":1126,"badge":1129,"source":1131},[1127],{"avatar":1128,"name":701,"to":702},{"src":700},{"label":1130},"Agent Security",{"name":706,"url":707},{"title":134,"description":1123},"dUKH-qn9TQKjrlSR7QlSBWtp7ntRfuCLlnB3xJxPkg0",1789550019217]