[{"data":1,"prerenderedAt":831},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Frobot-embodiment-interface":321,"\u002Fnews\u002Finsights\u002Frobot-embodiment-interface-surround":579},[4,8,17,21,25,29,33,309,313,317],{"title":5,"path":6,"stem":7},"About","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36,"children":37},"News","\u002Fnews","news",[38,41,65],{"title":39,"path":35,"stem":40},"News & Insights","news\u002Findex",{"title":18,"path":42,"stem":43,"children":44},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[45,49,53,57,61],{"title":46,"path":47,"stem":48},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":50,"path":51,"stem":52},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":54,"path":55,"stem":56},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":58,"path":59,"stem":60},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":62,"path":63,"stem":64},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":66,"stem":67,"children":68},"\u002Fnews\u002Finsights","news\u002Finsights",[69,73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297,301,305],{"title":70,"path":71,"stem":72},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":74,"path":75,"stem":76},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":78,"path":79,"stem":80},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":82,"path":83,"stem":84},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":86,"path":87,"stem":88},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":90,"path":91,"stem":92},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":94,"path":95,"stem":96},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":98,"path":99,"stem":100},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":102,"path":103,"stem":104},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":106,"path":107,"stem":108},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":110,"path":111,"stem":112},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":114,"path":115,"stem":116},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":118,"path":119,"stem":120},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":122,"path":123,"stem":124},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":126,"path":127,"stem":128},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":130,"path":131,"stem":132},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":134,"path":135,"stem":136},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":138,"path":139,"stem":140},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":142,"path":143,"stem":144},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":146,"path":147,"stem":148},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":150,"path":151,"stem":152},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":154,"path":155,"stem":156},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":158,"path":159,"stem":160},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":162,"path":163,"stem":164},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":166,"path":167,"stem":168},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":170,"path":171,"stem":172},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":174,"path":175,"stem":176},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":178,"path":179,"stem":180},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":182,"path":183,"stem":184},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":186,"path":187,"stem":188},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":190,"path":191,"stem":192},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":194,"path":195,"stem":196},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":198,"path":199,"stem":200},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":202,"path":203,"stem":204},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":206,"path":207,"stem":208},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":210,"path":211,"stem":212},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":214,"path":215,"stem":216},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":218,"path":219,"stem":220},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":222,"path":223,"stem":224},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":226,"path":227,"stem":228},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":230,"path":231,"stem":232},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":234,"path":235,"stem":236},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":238,"path":239,"stem":240},"Orchestration Without a Signal From the Room","\u002Fnews\u002Finsights\u002Fplanner-in-the-loop","news\u002Finsights\u002Fplanner-in-the-loop",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Body Was Supposed to Be the Easy Part","\u002Fnews\u002Finsights\u002Frobot-embodiment-interface","news\u002Finsights\u002Frobot-embodiment-interface",{"title":270,"path":271,"stem":272},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":274,"path":275,"stem":276},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":278,"path":279,"stem":280},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":282,"path":283,"stem":284},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":286,"path":287,"stem":288},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":290,"path":291,"stem":292},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":294,"path":295,"stem":296},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":298,"path":299,"stem":300},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":302,"path":303,"stem":304},"Algorithms Used in Autonomous Fighter Jet Flight Research","\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy","news\u002Finsights\u002Fvenom-f16-flight-autonomy",{"title":306,"path":307,"stem":308},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":310,"path":311,"stem":312},"Privacy Policy","\u002Fprivacy","privacy",{"title":314,"path":315,"stem":316},"Research","\u002Fresearch","research",{"title":318,"path":319,"stem":320},"Terms of Service","\u002Fterms","terms",{"id":322,"title":266,"body":323,"date":560,"description":561,"extension":562,"image":563,"meta":564,"navigation":576,"path":267,"seo":577,"stem":268,"__hash__":578},"insights\u002Fnews\u002Finsights\u002Frobot-embodiment-interface.md",{"type":324,"value":325,"toc":547},"minimark",[326,345,355,359,362,366,388,391,429,433,445,448,452,462,466,469],[327,328,331,332,331,338],"div",{"className":329},[330],"page-title","\n  ",[333,334,266],"h1",{"className":335,"id":337},[336],"page-title__main","the-body-was-supposed-to-be-the-easy-part",[339,340,344],"h2",{"className":341,"id":343},[342],"page-title__sub","every-humanoid-robot-has-needed-its-own-policy-trained-from-scratch-recent-research-asks-whether-a-shared-interface-can-finally-change-that","Every Humanoid Robot Has Needed Its Own Policy, Trained From Scratch. Recent Research Asks Whether a Shared Interface Can Finally Change That.",[327,346,348,349],{"style":347},"width: 100%; padding: 2%;","\n    ",[350,351],"img",{"src":352,"alt":353,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1769839271832-cfd7a1f6854f?w=1200&h=675&fit=crop&crop=focalpoint&fp-x=0.5&fp-y=0.32&auto=format","A humanoid robot's head, neck, and shoulders, wearing a blue lanyard","width: 100%; height: auto;",[356,357,358],"p",{},"A humanoid robot from one company does not share a skeleton with one from another. Arm lengths differ. Joint counts differ. Torque limits, sensor placement, and even how a knee is allowed to bend differ. For most of robotics' history, none of that has mattered much to the software riding on top, because a control policy, the trained model that decides what a robot's joints should do next, has been built for one specific machine and has stayed there. Retrain it, or even slightly change the robot's arm length or joint count, and the policy is often useless. It behaves less like software and more like a badge that only opens one particular door, no matter how many identical-looking doors get added down the hall.",[356,360,361],{},"That is the quieter problem sitting underneath the current wave of humanoid robot announcements. Several companies now ship their own distinct design, and a policy trained to walk one company's robot cannot walk another's, no matter how similar the two machines look standing still. What changed recently is a cluster of papers, working from different starting points, converging on the same fix, treating a robot's specific body the way a computer treats a specific chip, as a detail a shared interface should be able to hide rather than something a policy has to relearn from nothing.",[339,363,365],{"id":364},"one-body-learned-well","One Body, Learned Well",[356,367,368,369,373,374,382,383,387],{},"The first problem was more basic than making a policy work across many robots. It was making a single policy work well on one. Humanoid control has traditionally meant a library of separate, narrow skills, a walking gait here, a reaching motion there, each trained and tuned on its own. NVIDIA's SONIC controller, described in a paper published in ",[370,371,372],"em",{},"Science Robotics",", took a different approach, training one controller with roughly forty million parameters on a motion capture library far larger than earlier humanoid controllers had used, teaching it a broad vocabulary of human movement rather than a fixed set of tricks ",[375,376,377],"sup",{},[378,379,381],"a",{"href":380},"#source-1","[1]",". The result generalizes to movements it never specifically trained on, and it accepts commands from several different sources, virtual reality teleoperation, recorded video, or a separate vision language model, through what its authors describe as a single shared token space ",[375,384,385],{},[378,386,381],{"href":380},".",[356,389,390],{},"That shared token space is worth sitting with. It is not just a bigger model. It is a common vocabulary that different kinds of commands, whether from a human in a headset or another AI system giving instructions, get translated into before the robot's body ever has to interpret them. In computing terms, it functions like an instruction set, the fixed vocabulary of operations a chip agrees to support so that software written once can run without knowing exactly which transistors will execute it. SONIC built that instruction set for one specific robot body. The question left open was whether the same idea would survive contact with a different body entirely.",[327,392,348,394,348,398,348,404],{"style":393},"width: 100%; margin: 20px 0;",[350,395],{"src":396,"alt":397,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1564517945244-d371c925640b?w=1200&auto=format&fit=crop","Two different electrical adapters lying side by side on a plain background",[399,400,403],"h3",{"style":401,"id":402},"margin: 1rem 0 0.5rem 0;","any-body-same-instruction-set","Any Body, Same Instruction Set",[356,405,407,408,414,415,421,422,428],{"style":406},"margin: 0;","Swap the socket and the appliance still runs, because the plug was never the part doing the work. Robotics had already tested that logic once, on robot arms rather than legs, when the Open X-Embodiment collaboration pooled data from twenty-two different robots, contributed by research labs across several countries, and found that a policy trained on all of them transferred usefully to individual machines it had never specifically seen ",[375,409,410],{},[378,411,413],{"href":412},"#source-2","[2]",". Two more recent efforts pushed that same logic into humanoid whole-body control specifically. One system, built by researchers at Shanghai Jiao Tong University and the Shanghai Artificial Intelligence Laboratory, trains on a wide distribution of simulated robot shapes and physical properties so the resulting policy generalizes to real humanoids it never trained on directly, without any robot-specific retraining ",[375,416,417],{},[378,418,420],{"href":419},"#source-3","[3]",". A separate, newer framework goes a step further, explicitly separating the parts of a motion that come from shared human movement semantics, the timing and structure any body might share, from the parts that are specific to one robot's own physical execution, then routing each through a different piece of the model ",[375,423,424],{},[378,425,427],{"href":426},"#source-4","[4]",". Different research groups, different robot fleets, and the same underlying decision, that a body's particular geometry should sit behind the interface rather than baked into it.",[339,430,432],{"id":431},"what-the-interface-is-actually-made-of","What the Interface Is Actually Made Of",[356,434,435,436,440,441,387],{},"Saying a robot's body sits behind an interface is not yet an explanation. An interface only earns that name if something on one side of it stays fixed while something on the other side is free to change, and the harder design question in this line of research is exactly where that line gets drawn. The Shanghai team draws it before its model ever meets a specific robot, training the shared representation against a wide spread of simulated bodies and physical properties so that no single robot's proportions get baked into what the policy has learned to expect, while keeping the meaning of a given sensor reading or motor command consistent no matter which robot happens to be listening ",[375,437,438],{},[378,439,420],{"href":419},". The newer cross-embodiment framework draws the same line in a different place, splitting a human-centered vocabulary, tokens that describe a motion's shape and timing regardless of who or what is performing it, from a set of small, robot-specific modules whose only job is translating that shared vocabulary into one particular robot's own joints and proprioception ",[375,442,443],{},[378,444,427],{"href":426},[356,446,447],{},"What both designs have in common matters more than where they differ. The part that has to be rebuilt for a new robot is deliberately kept small, and almost everything the policy actually knows about movement lives on the shared side of that line. That allocation is the entire point of building an interface in the first place. A translation layer that has to relearn most of what it knows every time a new robot shows up is not really an interface, it is just a second training run with extra steps.",[339,449,451],{"id":450},"what-the-interface-does-not-settle","What the Interface Does Not Settle",[356,453,454,455,461],{},"None of this closes the distance between a research demonstration and a robot working somewhere messier than a motion capture stage or a lab floor, and that gap is worth stating plainly rather than waving past. Even a single, well-behaved robot body running in a clean simulator does not automatically match how that same body moves once it is welded, motored, and standing on real ground, a mismatch specific enough that an earlier project needed a dedicated second training stage just to correct for it after the fact ",[375,456,457],{},[378,458,460],{"href":459},"#source-5","[5]",". Extending that correction from one lab robot to a policy meant to generalize across many bodies, then further out to homes, warehouses, or outdoor terrain nobody trained on, is a considerably larger claim than anything in the papers above actually tested. What the interface research settles is narrower and, in its own way, more useful, whether a body has to be relearned from scratch every time it changes shape.",[339,463,465],{"id":464},"what-this-suggests","What This Suggests",[356,467,468],{},"Every new humanoid robot announcement invites the same question, whether this particular machine can actually do the job asked of it. The research underneath that question suggests the frontier is not really about any one body. It is about whether a shared interface, a vocabulary of motion that means the same thing regardless of which machine is listening, can keep holding as more and increasingly different bodies get added behind it. Computing went through a comparable split decades ago, separating what a chip is from what a program written for it has to know, and that separation took a long, uneven stretch of competing designs before it became something the field simply assumed rather than argued about. Robotics looks to be partway through an equivalent shift now, still testing how far one interface can stretch before a new robot's proportions or actuators break the abstraction, and how much of the underlying skill survives that stretch intact.",[327,470,331,474,331,477],{"className":471},[472,473],"references","mt-8",[339,475,476],{"id":472},"References",[478,479,348,485,348,501,348,513,348,525,348,535,331],"ol",{"className":480},[481,482,483,484],"list-decimal","list-inside","space-y-2","mt-4",[486,487,489,490,492,493],"li",{"id":488},"source-1","Z. Luo et al., \"Supersizing Motion Tracking for Natural Humanoid Whole-Body Control,\" ",[370,491,372],{},", vol. 11, no. 117, eaed4592, 2026. DOI: ",[378,494,500],{"href":495,"target":496,"className":497},"https:\u002F\u002Fdoi.org\u002F10.1126\u002Fscirobotics.aed4592","_blank",[498,499],"text-blue-600","underline","[Online]",[486,502,504,505,508,509],{"id":503},"source-2","A. O'Neill et al., \"Open X-Embodiment: Robotic Learning Datasets and RT-X Models,\" in ",[370,506,507],{},"Proc. IEEE International Conference on Robotics and Automation (ICRA)",", 2024, pp. 6892-6903. DOI: ",[378,510,500],{"href":511,"target":496,"className":512},"https:\u002F\u002Fdoi.org\u002F10.1109\u002FICRA57147.2024.10611477",[498,499],[486,514,516,517,520,521],{"id":515},"source-3","Y. Xue et al., \"Scalable and General Whole-Body Control for Cross-Humanoid Locomotion,\" ",[370,518,519],{},"arXiv",", 2026, ",[378,522,500],{"href":523,"target":496,"className":524},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2602.05791",[498,499],[486,526,528,529,520,531],{"id":527},"source-4","J. Zhang et al., \"X-WBC: A Cross-Embodiment Foundation Model for Humanoid Whole-Body Control,\" ",[370,530,519],{},[378,532,500],{"href":533,"target":496,"className":534},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.15213",[498,499],[486,536,538,539,542,543],{"id":537},"source-5","T. He et al., \"ASAP: Aligning Simulation and Real-World Physics for Learning Agile Humanoid Whole-Body Skills,\" in ",[370,540,541],{},"Proc. Robotics: Science and Systems (RSS)",", 2025. DOI: ",[378,544,500],{"href":545,"target":496,"className":546},"https:\u002F\u002Fdoi.org\u002F10.15607\u002FRSS.2025.XXI.066",[498,499],{"title":548,"searchDepth":549,"depth":549,"links":550},"",2,[551,552,556,557,558,559],{"id":343,"depth":549,"text":344},{"id":364,"depth":549,"text":365,"children":553},[554],{"id":402,"depth":555,"text":403},3,{"id":431,"depth":549,"text":432},{"id":450,"depth":549,"text":451},{"id":464,"depth":549,"text":465},{"id":472,"depth":549,"text":476},"2026-09-14","Humanoid robots from different companies do not share a skeleton, and until recently that meant a control policy built for one robot could not walk another. A cluster of new research treats a robot's body as a detail a shared interface should hide, not something a policy has to relearn from scratch every time it changes shape.","md",{"src":352},{"authors":565,"badge":571,"source":573},[566],{"avatar":567,"name":569,"to":570},{"src":568},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":572},"Embodied AI",{"name":574,"url":575},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":266,"description":561},"ksvAsVQopvVcFzVTMVGHQNVNPM9rKnmRVRvmK7p0RPk",[580,581],null,{"id":582,"title":238,"body":583,"date":819,"description":820,"extension":562,"image":821,"meta":822,"navigation":576,"path":239,"seo":829,"stem":240,"__hash__":830,"_path":239},"insights\u002Fnews\u002Finsights\u002Fplanner-in-the-loop.md",{"type":324,"value":584,"toc":807},[585,597,603,606,609,612,616,624,637,640,644,647,655,663,666,670,678,686,698,712,720,724,727,736,739,742],[327,586,331,588,331,592],{"className":587},[330],[333,589,238],{"className":590,"id":591},[336],"orchestration-without-a-signal-from-the-room",[339,593,596],{"className":594,"id":595},[342],"agent-graphs-can-call-tools-the-conductor-often-does-not-hear-whether-the-job-finished","Agent Graphs Can Call Tools. The Conductor Often Does Not Hear Whether the Job Finished.",[327,598,348,599],{"style":347},[350,600],{"src":601,"alt":602,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1745328599097-5b2adfa0087e?w=1200&auto=format&fit=crop","A conductor facing an orchestra during rehearsal, listening rather than reading a score in isolation",[356,604,605],{},"A conductor who has only ever studied the score is not the same as a conductor who has stood in front of that particular orchestra. The score already knows which entrance belongs to the horns. What it cannot tell anyone is whether those horns will actually arrive on time tonight, or whether the first violin is sitting a half-beat late, or whether the room itself is swallowing the low strings. Those facts only show up in rehearsal.",[356,607,608],{},"Most production agent stacks are still closer to the score. A large language model, a system trained to generate text by predicting the next word, gets wrapped in a graph of specialists with names like planner, executor, critic, and coder. Each node is a frozen model plus a prompt. Tools get called. Memory gets appended. The graph looks like architecture. Orchestration, the work of deciding what happens next, which tool, which sub-goal, whether the last result was even usable, often does not receive a learning signal from the trajectory it just caused. The room already answered. The graph did not listen.",[356,610,611],{},"From a systems perspective, that missing wire is the whole story. A deployed controller that cannot hear the plant will keep conducting from last week's diagram. Prompt edits after a surprising failure are not the same as a gradient, the training signal that tells a model how to change. They are a human rewriting the score by hand.",[339,613,615],{"id":614},"a-score-with-every-entrance-marked","A Score With Every Entrance Marked",[356,617,618,619,623],{},"The conversation frameworks that made agent graphs easy to ship did something useful. They made orchestration programmable as talk among specialists, humans, and tools, without requiring a training loop for that particular collaboration ",[375,620,621],{},[378,622,381],{"href":380},". The graph became a product surface. Roles could be named. Handoffs could be coded. What those designs did not attach, by construction, was a learning signal to the talk itself. When a search result was noisy, a code tool returned an exception, or an early sub-goal was the wrong one, the usual fix was another prompt. Prompting problems tend to be rewritten after every surprising failure.",[327,625,348,626,348,630,348,634],{"style":393},[350,627],{"src":628,"alt":629,"style":354},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1750449821191-68a5facd2be7?w=1200&auto=format&fit=crop","A vintage control room of labeled switches and analog panels that stay in the same positions",[399,631,633],{"style":401,"id":632},"every-switch-already-has-a-label","Every Switch Already Has a Label",[356,635,636],{"style":406},"A room like this looks finished. Every circuit has a nameplate and a home position. What it does not have is a way for the person at the board to get better at using it from the last shift's near misses. Frozen agent graphs have the same finished look. The planner, executor, and verifier all have roles. In the default design, none of those roles takes a gradient from what the last tool call actually returned.",[356,638,639],{},"That is orchestration without a signal from the room. It can still run a job. It cannot, on its own, get better at sequencing the job from the job's own outcome.",[339,641,643],{"id":642},"a-transcript-is-not-the-room-either","A Transcript Is Not the Room Either",[356,645,646],{},"The obvious patch is to train the orchestrator after all, just not in the live loop. Collect traces from a stronger model, copy them token by token, and call the result learning. The traces look like rehearsal notes. They are still a score. They were written for a different night, a different band, a different set of missed entrances.",[356,648,649,650,654],{},"Work on in-the-flow agent training put that patch to a direct test. Inside the same modular graph, a planner, an executor, a verifier, and a generator sharing a structured memory, distilling a stronger model's planner traces into a smaller planner produced a collapse, a 19 percent drop in average accuracy relative to leaving that planner frozen ",[375,651,652],{},[378,653,413],{"href":412},". The imitation objective was answering the wrong question. It rewarded looking like a good transcript, not finishing the work after a tool had already gone sideways.",[356,656,657,658,662],{},"The same graph, with the planner updated from the live rollouts it actually induced, recovered that ground. On-policy training, updates from the trajectories the current system produces rather than from a cleaned-up archive, is what puts the orchestrator back in the room. The planner still sees the live state, the query, the toolset, the current memory, because that is the state it will face at inference. After that training, tool choice moved with the task instead of staying at a default search habit. Broad factual questions pulled more web search. A medical question set spent more of the budget inside Wikipedia and page-level retrieval ",[375,659,660],{},[378,661,413],{"href":412},". A frozen prompt is supposed to produce that kind of shift and often does not, because the prompt cannot see, in gradient terms, which of those choices ended in a correct answer.",[356,664,665],{},"The lesson is narrower than \"train a planner.\" It is about which signal counts. A transcript is a record of how someone else conducted a different night. A live outcome is what this orchestra, in this room, actually did.",[339,667,669],{"id":668},"dumping-the-outcome-through-the-whole-band","Dumping the Outcome Through the Whole Band",[356,671,672,673,677],{},"Some stacks do send a final correct-or-incorrect score back through the model that issued the tool calls. The search engine sits in the environment. The model learns when to query and how to use what comes back, with retrieved tokens masked so the update does not try to credit the search engine's text as if the model had written it ",[375,674,675],{},[378,676,420],{"href":419},". That is a real signal. It is also a signal dumped through every thought, every tool choice, and every wording decision in one long context.",[356,679,680,681,685],{},"The room is noisy. Tool output is often off the pretrained distribution. Even when those tokens are masked from the loss, the model's next generation inherits the shift, samples increasingly unlikely tokens, and the gradient can explode. One practical response has been to throw out entire trajectories that contain a void turn, a response that produced neither a code block nor a final answer, because those turns were pulling the update in the wrong direction ",[375,682,683],{},[378,684,427],{"href":426},". The filter is a way of saying the room spoke, and some of what it said was not a usable learning signal.",[356,687,688,689,693,694,697],{},"A related failure shows up when the only reward is that the episode worked. Reward variability collapses, gradients spike, and the agent starts repeating locally rewarded patterns that do not amount to reasoning. Shallow strategies and hallucinated thoughts are easy to grow if the signal has nothing to say about whether the thoughts were real ",[375,690,691],{},[378,692,460],{"href":459},". That sits next to an ",[378,695,696],{"href":259},"earlier Thinkata look at temporal credit assignment",". A single terminal score is a hard thing to send backward through a long chain. Stuffing it through the entire policy makes the chain even longer.",[327,699,348,700,348,705,348,709],{"style":393},[350,701],{"src":702,"alt":703,"style":704},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1775519519950-9c48495b5488?w=1200&h=1200&fit=crop&crop=focalpoint&fp-x=0.52&fp-y=0.38&auto=format","An industrial panel of gauges and switches where only some circuits are live","width: 100%; aspect-ratio: 1 \u002F 1; object-fit: cover;",[399,706,708],{"style":401,"id":707},"the-wire-has-to-reach-the-loop-that-chooses","The Wire Has to Reach the Loop That Chooses",[356,710,711],{"style":406},"A working plant does not rewire every gauge because one loop is drifting. It retunes the controller that actually sets the next action. The learning signal has to land somewhere specific. Spread it across every token in a growing context and the noise in the room can swamp the update. Leave it unconnected and the labeled panel does not learn from the last shift.",[356,713,714,715,719],{},"The engineering question is not whether a terminal outcome is too crude. Outcome rewards are often the only score that can be checked. The question is where that crude score is allowed to change the system. One measured answer has been to keep the modular graph and copy a single verifiable result, right or wrong at the end, onto every orchestrator turn, while each turn still conditions on the full memory so far ",[375,716,717],{},[378,718,413],{"href":412},". The local decision is not blind. The update still points at whether the job finished. That is a systems choice about which module is allowed to hear the room, not a new theory of credit.",[339,721,723],{"id":722},"how-much-of-the-room-needs-to-hear-it","How Much of the Room Needs to Hear It",[356,725,726],{},"Training only the conductor is a bet about the rest of the stack. Tools, executors, and verifiers can stay versioned infrastructure. When the toolset or the task mix changes, the thing that gets retrained is the loop that chooses, not the whole orchestra. That matches how a lot of production software is already pinned.",[356,728,729,730,387],{},"It is also incomplete in a specific way. If the models that actually run the steps stay frozen while only the designer or planner learns, part of the loop is still conducting from a score. Work on automatic multi-agent systems has started treating that gap as a ceiling. Jointly training the side that writes the workflow and the side that executes it produced gains that leaving either side frozen did not, and the two roles appeared to improve on a staggered schedule rather than all at once ",[375,731,732],{},[378,733,735],{"href":734},"#source-6","[6]",[356,737,738],{},"That does not settle how much of an agent graph should be trainable. For a team whose executor is a deterministic API, a search endpoint, a code runner, a pinned verifier, putting the learning signal on the orchestrator is the operable experiment. For a team whose executors are themselves language models with their own failure modes, freezing them may be the part of the room that later looks deaf. Domain practitioners will know better than a systems argument whether a given executor is a tool or a second conductor.",[356,740,741],{},"What does seem worth taking seriously is the missing wire itself. A prompt-orchestrated graph is a practical way to run specialists. It does not, by itself, teach the module that sequences them. Sending the outcome through every token of a monolithic policy can teach that sequencing, and then spend the training budget on not exploding. The option that matches the rest of a production stack is smaller. Pin the tools. Pin the memory format. Attach a learning signal to the loop that is actually conducting, from the trajectory that loop just caused, in the room it is standing in tonight.",[327,743,331,745,331,747],{"className":744},[472,473],[339,746,476],{"id":472},[478,748,348,750,348,760,348,770,348,779,348,788,348,797,331],{"className":749},[481,482,483,484],[486,751,752,753,755,756],{"id":488},"Q. Wu et al., \"AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation,\" ",[370,754,519],{},", 2023, ",[378,757,500],{"href":758,"target":496,"className":759},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2308.08155",[498,499],[486,761,762,763,765,766],{"id":503},"Z. Li et al., \"In-the-Flow Agentic System Optimization for Effective Planning and Tool Use,\" ",[370,764,519],{},", 2025, ",[378,767,500],{"href":768,"target":496,"className":769},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.05592",[498,499],[486,771,772,773,765,775],{"id":515},"B. Jin et al., \"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning,\" ",[370,774,519],{},[378,776,500],{"href":777,"target":496,"className":778},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2503.09516",[498,499],[486,780,781,782,765,784],{"id":527},"Z. Xue et al., \"SimpleTIR: End-to-End Reinforcement Learning for Multi-Turn Tool-Integrated Reasoning,\" ",[370,783,519],{},[378,785,500],{"href":786,"target":496,"className":787},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.02479",[498,499],[486,789,790,791,765,793],{"id":537},"Z. Wang et al., \"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning,\" ",[370,792,519],{},[378,794,500],{"href":795,"target":496,"className":796},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2504.20073",[498,499],[486,798,800,801,520,803],{"id":799},"source-6","Y. Zhang et al., \"MetaAgent-X: Breaking the Ceiling of Automatic Multi-Agent Systems via End-to-End Reinforcement Learning,\" ",[370,802,519],{},[378,804,500],{"href":805,"target":496,"className":806},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.14212",[498,499],{"title":548,"searchDepth":549,"depth":549,"links":808},[809,810,813,814,817,818],{"id":595,"depth":549,"text":596},{"id":614,"depth":549,"text":615,"children":811},[812],{"id":632,"depth":555,"text":633},{"id":642,"depth":549,"text":643},{"id":668,"depth":549,"text":669,"children":815},[816],{"id":707,"depth":555,"text":708},{"id":722,"depth":549,"text":723},{"id":472,"depth":549,"text":476},"2026-09-07","Agent stacks can look like architecture, specialists, tools, memory, a planner, and still send no training signal back to the module that is conducting the work. The room already answered. The graph often does not listen.",{"src":601},{"authors":823,"badge":826,"source":828},[824],{"avatar":825,"name":569,"to":570},{"src":568},{"label":827},"Agent Systems",{"name":574,"url":575},{"title":238,"description":820},"0Kpu6a_IGPAxywvnv3uk3v9OZc-Tg293WrjvZc_kotI",1789550019217]