[{"data":1,"prerenderedAt":939},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Finterleaved-distillation-rl":329,"\u002Fnews\u002Finsights\u002Finterleaved-distillation-rl-surround":731},[4,8,17,21,25,29,33,317,321,325],{"title":5,"path":6,"stem":7},"About","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36,"children":37},"News","\u002Fnews","news",[38,41,65],{"title":39,"path":35,"stem":40},"News & Insights","news\u002Findex",{"title":18,"path":42,"stem":43,"children":44},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[45,49,53,57,61],{"title":46,"path":47,"stem":48},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":50,"path":51,"stem":52},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":54,"path":55,"stem":56},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":58,"path":59,"stem":60},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":62,"path":63,"stem":64},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":66,"stem":67,"children":68},"\u002Fnews\u002Finsights","news\u002Finsights",[69,73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297,301,305,309,313],{"title":70,"path":71,"stem":72},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":74,"path":75,"stem":76},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":78,"path":79,"stem":80},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":82,"path":83,"stem":84},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":86,"path":87,"stem":88},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":90,"path":91,"stem":92},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":94,"path":95,"stem":96},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":98,"path":99,"stem":100},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":102,"path":103,"stem":104},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":106,"path":107,"stem":108},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":110,"path":111,"stem":112},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":114,"path":115,"stem":116},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":118,"path":119,"stem":120},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":122,"path":123,"stem":124},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":126,"path":127,"stem":128},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":130,"path":131,"stem":132},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":134,"path":135,"stem":136},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":138,"path":139,"stem":140},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":142,"path":143,"stem":144},"A Panel Can Be Safer Than Any Member of It","\u002Fnews\u002Finsights\u002Fcoalitional-alignment-panel-review","news\u002Finsights\u002Fcoalitional-alignment-panel-review",{"title":146,"path":147,"stem":148},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":150,"path":151,"stem":152},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":154,"path":155,"stem":156},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":158,"path":159,"stem":160},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":162,"path":163,"stem":164},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":166,"path":167,"stem":168},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":170,"path":171,"stem":172},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":174,"path":175,"stem":176},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":178,"path":179,"stem":180},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":182,"path":183,"stem":184},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":186,"path":187,"stem":188},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":190,"path":191,"stem":192},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":194,"path":195,"stem":196},"Taking Turns Instead of Splitting the Difference","\u002Fnews\u002Finsights\u002Finterleaved-distillation-rl","news\u002Finsights\u002Finterleaved-distillation-rl",{"title":198,"path":199,"stem":200},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":202,"path":203,"stem":204},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":206,"path":207,"stem":208},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":210,"path":211,"stem":212},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":214,"path":215,"stem":216},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":218,"path":219,"stem":220},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":222,"path":223,"stem":224},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":226,"path":227,"stem":228},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":230,"path":231,"stem":232},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":234,"path":235,"stem":236},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":238,"path":239,"stem":240},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":242,"path":243,"stem":244},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":246,"path":247,"stem":248},"Orchestration Without a Signal From the Room","\u002Fnews\u002Finsights\u002Fplanner-in-the-loop","news\u002Finsights\u002Fplanner-in-the-loop",{"title":250,"path":251,"stem":252},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":254,"path":255,"stem":256},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":258,"path":259,"stem":260},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":262,"path":263,"stem":264},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":266,"path":267,"stem":268},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":270,"path":271,"stem":272},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":274,"path":275,"stem":276},"The Body Was Supposed to Be the Easy Part","\u002Fnews\u002Finsights\u002Frobot-embodiment-interface","news\u002Finsights\u002Frobot-embodiment-interface",{"title":278,"path":279,"stem":280},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":282,"path":283,"stem":284},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":286,"path":287,"stem":288},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":290,"path":291,"stem":292},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":294,"path":295,"stem":296},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":298,"path":299,"stem":300},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":302,"path":303,"stem":304},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":306,"path":307,"stem":308},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":310,"path":311,"stem":312},"Algorithms Used in Autonomous Fighter Jet Flight Research","\u002Fnews\u002Finsights\u002Fvenom-f16-flight-autonomy","news\u002Finsights\u002Fvenom-f16-flight-autonomy",{"title":314,"path":315,"stem":316},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":318,"path":319,"stem":320},"Privacy Policy","\u002Fprivacy","privacy",{"title":322,"path":323,"stem":324},"Research","\u002Fresearch","research",{"title":326,"path":327,"stem":328},"Terms of Service","\u002Fterms","terms",{"id":330,"title":194,"body":331,"date":712,"description":713,"extension":714,"image":715,"meta":716,"navigation":728,"path":195,"seo":729,"stem":196,"__hash__":730},"insights\u002Fnews\u002Finsights\u002Finterleaved-distillation-rl.md",{"type":332,"value":333,"toc":698},"minimark",[334,353,363,367,370,382,386,400,410,418,422,430,440,448,458,466,470,478,486,501,505,517,539,561,564,568,571,579,587],[335,336,339,340,339,346],"div",{"className":337},[338],"page-title","\n  ",[341,342,194],"h1",{"className":343,"id":345},[344],"page-title__main","taking-turns-instead-of-splitting-the-difference",[347,348,352],"h2",{"className":349,"id":351},[350],"page-title__sub","why-reinforcement-learning-and-distillation-may-work-better-in-alternation-than-in-a-single-blended-loss","Why Reinforcement Learning and Distillation May Work Better in Alternation Than in a Single Blended Loss",[335,354,356,357],{"style":355},"width: 100%; padding: 2%;","\n    ",[358,359],"img",{"src":360,"alt":361,"style":362},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1772517403501-c4bb361a4bdc?w=1200&h=750&fit=crop&auto=format","Two teams leaning hard in opposite directions on a taut rope across a grass field","width: 100%; height: auto;",[364,365,366],"p",{},"Both teams in a tug of war can be pulling as hard as they can while the rope barely moves. All of that effort is real, and most of it cancels out.",[364,368,369],{},"Training a large language model after pretraining, a stage usually called post-training, often involves two forces that behave in a similar way. One is reinforcement learning, where the model attempts a problem, receives a reward based on how well the answer checks out, and becomes more likely to produce whatever earned that reward. The other is distillation, where a smaller model, the student, learns by imitating a larger and more capable model, the teacher. Both are widely used to improve models. They also pull on the same property of the model in opposite directions. One straightforward way to combine them is to fold both objectives into a single weighted loss, the score training tries to push down, so that one update carries both.",[364,371,372,373,381],{},"A technical report from ByteDance describing Pistis, a family of multimodal models that read images and video as well as text, makes a case for a different arrangement ",[374,375,376],"sup",{},[377,378,380],"a",{"href":379},"#source-1","[1]",". Instead of blending the two signals, its training loop alternates between them. The measured gains are modest and self-reported. The schedule illustrates a broader problem that can arise whenever one model answers to competing objectives, and two other parts of the report show a similar preference for keeping signals apart, one in the credit given to an agent's individual actions and one in the software wrapped around a model that is no longer being trained.",[347,383,385],{"id":384},"two-signals-that-pull-apart","Two Signals That Pull Apart",[364,387,388,389,395,396,399],{},"The property in question is entropy, a measure of how spread out a model's choices are. A model with high entropy keeps many possible next words in play at each step. A model with low entropy commits hard to one or two. Reinforcement learning tends to drive entropy down, because rewarding a correct answer raises the probability of the exact words that produced it. A study of reinforcement learning on reasoning models found that without deliberate intervention, entropy falls sharply early in training, and performance flattens once that spread has been used up ",[374,390,391],{},[377,392,394],{"href":393},"#source-2","[2]",". Much of the drop came from words the model already favored that also earned a high reward, which then crowd out the alternatives. An earlier Thinkata piece covered ",[377,397,398],{"href":127},"entropy collapse"," in classic control settings.",[364,401,402,403,409],{},"On-policy distillation works differently. The student writes its own answer, and the teacher then scores every word of it, reporting what it would have chosen at that same point ",[374,404,405],{},[377,406,408],{"href":407},"#source-3","[3]",". Training on its own attempts means the student gets corrected in the situations it actually wanders into. The signal is also dense, a target at every word rather than a single pass or fail at the end.",[364,411,412,413,417],{},"Depending on how the gap between student and teacher is measured, distillation can either narrow the student's choices or widen them. Pistis uses Jensen-Shannon divergence, a measure of how far apart two probability distributions are, computed over the teacher's fifty most likely next words. That setup asks the student to cover the range of options the teacher considers plausible ",[374,414,415],{},[377,416,380],{"href":379},". In the team's comparison, a version restricted to the top five words still let entropy collapse, and so did a popular alternative that compares the two models only on the single word the student actually wrote. Only the broad version held entropy up.",[347,419,421],{"id":420},"why-the-sum-goes-wrong","Why the Sum Goes Wrong",[364,423,424,425,429],{},"When both objectives are folded into one loss, every training step produces one combined direction in which to move the model's weights. The trouble comes on words where the student is already more confident than the teacher. If that word belonged to a rewarded answer, reinforcement learning pushes its probability higher still, while distillation pushes it back down toward the teacher's level. A first-order analysis in the report finds that the disagreement between the two directions subtracts from the progress each would make on its own ",[374,426,427],{},[377,428,380],{"href":379},". Past a certain degree of conflict, the blended step can actually lower the reward even while the combined loss improves.",[364,431,432,433,439],{},"Multi-task learning hit a version of this years earlier. A study of networks trained on several tasks at once identified conflicting gradients, the per-task directions of improvement pointing against one another, as one of a small set of conditions behind detrimental gradient interference ",[374,434,435],{},[377,436,438],{"href":437},"#source-4","[4]",". Their fix, which they call gradient surgery, edits each gradient, stripping out the part that fights the other before combining them.",[364,441,442,443,447],{},"Pistis takes a blunter route. Its schedule, called Interleaved Distillation and Reinforcement Learning, or IDRL, runs five steps of distillation, then five steps of reinforcement learning, and repeats ",[374,444,445],{},[377,446,380],{"href":379},". Each individual step answers to one objective only, so neither can cancel the other within an update. The two still shape each other, just across phases rather than inside a single gradient.",[364,449,450,451,457],{},"The distillation phases play a role similar to the penalty many reinforcement learning setups apply to keep a model from drifting too far from where it started, a connection a recent survey of on-policy distillation lays out formally ",[374,452,453],{},[377,454,456],{"href":455},"#source-5","[5]",". Here the anchor is a stronger model rather than an earlier copy of the student, so the pull toward it can add ability instead of only restraining change. It also sets a ceiling, since a student constantly drawn toward its teacher cannot pass it.",[364,459,460,461,465],{},"Once entropy stops rising, the schedule drops distillation and continues with reinforcement learning alone ",[374,462,463],{},[377,464,380],{"href":379},". Only the smaller 9-billion-parameter models were trained this way. The 27-billion-parameter models used reinforcement learning alone, then served as the frozen teachers.",[347,467,469],{"id":468},"what-the-comparison-actually-shows","What the Comparison Actually Shows",[364,471,472,473,477],{},"From the same fine-tuned checkpoint, the team trained its 9-billion-parameter agent model five ways, with reinforcement learning alone, distillation alone, distillation followed by reinforcement learning, both objectives summed at every step, and the alternating schedule. Across eighteen benchmarks spanning charts, visual reasoning, search, and tool use, alternation came out on top ",[374,474,475],{},[377,476,380],{"href":379},". The summed loss and the two-stage handoff tied just behind it, a bit more than half a point lower on the averaged score. That comparison ran on one model size with one teacher, and the report shows no error bars or repeated training runs, so it cannot say whether a gap that small would survive a rerun. The authors say the results support interleaving without establishing that reduced gradient conflict or better exploration explains the difference.",[364,479,480,481,485],{},"The training curves tell a clearer story than the scores. Reinforcement learning on its own showed the steady entropy decline the earlier research predicts, along with gradient norms that climbed and spiked late in training, a sign of unstable updates. The variants that mixed in distillation throughout kept entropy higher, and the alternating run traced a sawtooth, entropy dipping in each reinforcement phase and recovering in each distillation phase ",[374,482,483],{},[377,484,380],{"href":379},". Search tasks were where alternation pulled furthest ahead of the single handoff.",[364,487,488,489,493,494,500],{},"Distillation also turned out to have a precondition. Applied directly to a student that had not been warmed up, it barely moved accuracy on a geometry benchmark, and its training curve dipped sharply at the start ",[374,490,491],{},[377,492,380],{"href":379},". A supervised phase on teacher-written answers, run first, fixed that. A separate study of text-only models found that on-policy distillation depends on student and teacher sharing compatible patterns of reasoning, and that this kind of cold start can rescue failing runs ",[374,495,496],{},[377,497,499],{"href":498},"#source-6","[6]",". The Pistis results suggest the same holds once images enter the picture.",[347,502,504],{"id":503},"the-same-separation-outside-the-gradient","The Same Separation Outside the Gradient",[364,506,507,508,511,512,516],{},"The report shows a similar preference in its handling of credit assignment, the problem of deciding which actions in a long sequence deserve credit for the final outcome, also the subject of an earlier Thinkata piece on ",[377,509,510],{"href":267},"temporal credit assignment",". When an agent searches over many turns and lands on the right answer, a reward attached only to that answer credits every step along the way, including malformed tool calls, repeated or near-duplicate calls, calls made after the model was told to answer, and calls whose results could not be fed back into its context. Pistis zeroes out positive credit for those steps while leaving any penalty in place ",[374,513,514],{},[377,515,380],{"href":379},". Removing that rule hurt the search tasks most and barely changed short tasks like chart reading, which fits the idea that noisy credit compounds with length.",[335,518,356,520,356,524,356,530],{"style":519},"width: 100%; margin: 20px 0;",[358,521],{"src":522,"alt":523,"style":362},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1462642109801-4ac2971a3a51?w=1200&auto=format&fit=crop","An open notebook with a numbered handwritten entry and short notes indented beneath it, a fountain pen resting on the page",[525,526,529],"h3",{"style":527,"id":528},"margin: 1rem 0 0.5rem 0;","a-ledger-around-a-frozen-model","A Ledger Around a Frozen Model",[364,531,533,534,538],{"style":532},"margin: 0;","A numbered entry with a few short notes indented beneath it is about as plain as record keeping gets, and the second half of the report builds something close to it. Pistis-Auto-Harnessing, or PAH, leaves the trained model untouched and instead improves its harness, the surrounding software that decides what the model sees, which tools it can call, and when it has to stop and answer ",[374,535,536],{},[377,537,380],{"href":379},". During development, a separate optimization agent reads failed runs, proposes one reversible change at a time, and keeps it only if a fixed development score improves. The test set is used once, after the harness is frozen. The harness that came out of this process keeps a running ledger of candidate answers, each tied to the evidence for and against it and to the conditions it has not yet satisfied, while the model itself still makes every decision.",[364,540,541,542,548,549,555,556,560],{},"Related work has pursued the same target more broadly. One line of research uses a meta agent that writes new agent designs in code ",[374,543,544],{},[377,545,547],{"href":546},"#source-7","[7]",", and another evolves prompts by reflecting in natural language on complete runs ",[374,550,551],{},[377,552,554],{"href":553},"#source-8","[8]",". PAH is narrower by design. On a multimodal search benchmark, with every attempt held to the same budget of fifteen search and page-reading actions, the optimized harness got about nine more questions right out of five hundred, and the average number of actions barely moved ",[374,557,558],{},[377,559,380],{"href":379},". Placed around a different model, GPT-5.5, the same frozen harness helped by a wider margin. Transfer outside multimodal search has not been tested.",[364,562,563],{},"The three mechanisms work for different reasons. Interleaving changes when each gradient applies, the credit rule changes which actions a reward reaches, and the harness loop moves part of the search out of the weights entirely. What they share is a design preference rather than a common mechanism, a habit of keeping apart signals that a simpler design would merge.",[347,565,567],{"id":566},"what-remains-uncertain","What Remains Uncertain",[364,569,570],{},"How far this generalizes is still open. The key IDRL and PAH results come from the team's own experiments, and the preprint, posted only days ago, has not yet been independently reproduced or peer reviewed. With no sweep over phase lengths, it is hard to tell whether the benefit depends on the ratio, the length, or simply alternating. A teacher from another model family, or a much smaller student, might also change the picture, given how sensitive distillation appears to be to the starting overlap between the two.",[364,572,573,574,578],{},"The report's own failure analysis points somewhere else. Even the improved agent sometimes wrote code that confirmed a guess instead of measuring anything, or found the right fact during search and then contradicted it in its final answer ",[374,575,576],{},[377,577,380],{"href":379},". Neither failure is obviously addressed by entropy management or gradient separation. Both concern whether intermediate steps actually constrain the outcome, which suggests the next useful reward may need to score the process rather than the answer.",[364,580,581,582,586],{},"For teams that fold a distillation term and a reward term into one loss, the report offers a useful diagnostic more than a new recipe. Logging the agreement between the two gradients over a run, often measured as the cosine of the angle between them, would show whether they point against each other at all, though computing them separately adds real cost at large scale. Opposition alone is not proof of harm. The multi-task study behind gradient surgery found conflicting gradients damaging mainly when the loss surface also curves sharply and one gradient is much larger than the other ",[374,583,584],{},[377,585,438],{"href":437},". If strong opposition is absent, gradient conflict becomes a less convincing explanation for any weakness in the blended loss. If it grows as the model becomes more confident, alternating the two may be worth trying before retuning the weights on the sum.",[335,588,339,592,339,595],{"className":589},[590,591],"references","mt-8",[347,593,594],{"id":590},"References",[596,597,356,603,356,621,356,632,356,644,356,656,356,666,356,676,356,687,339],"ol",{"className":598},[599,600,601,602],"list-decimal","list-inside","space-y-2","mt-4",[604,605,607,608,612,613],"li",{"id":606},"source-1","H. Chen et al., \"Pistis Technical Report,\" ",[609,610,611],"em",{},"arXiv",", 2026, ",[377,614,620],{"href":615,"target":616,"className":617},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.28554","_blank",[618,619],"text-blue-600","underline","[Online]",[604,622,624,625,627,628],{"id":623},"source-2","G. Cui et al., \"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models,\" ",[609,626,611],{},", 2025, ",[377,629,620],{"href":630,"target":616,"className":631},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2505.22617",[618,619],[604,633,635,636,639,640],{"id":634},"source-3","R. Agarwal et al., \"On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes,\" in ",[609,637,638],{},"Proc. International Conference on Learning Representations (ICLR'24)",", 2024, ",[377,641,620],{"href":642,"target":616,"className":643},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2306.13649",[618,619],[604,645,647,648,651,652],{"id":646},"source-4","T. Yu et al., \"Gradient Surgery for Multi-Task Learning,\" in ",[609,649,650],{},"Advances in Neural Information Processing Systems",", vol. 33, 2020, ",[377,653,620],{"href":654,"target":616,"className":655},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2001.06782",[618,619],[604,657,659,660,612,662],{"id":658},"source-5","M. Song and M. Zheng, \"A Survey of On-Policy Distillation for Large Language Models,\" ",[609,661,611],{},[377,663,620],{"href":664,"target":616,"className":665},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.00626",[618,619],[604,667,669,670,612,672],{"id":668},"source-6","Y. Li et al., \"Rethinking On-Policy Distillation of Large Language Models: Phenomenology, Mechanism, and Recipe,\" ",[609,671,611],{},[377,673,620],{"href":674,"target":616,"className":675},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.13016",[618,619],[604,677,679,680,627,683],{"id":678},"source-7","S. Hu et al., \"Automated Design of Agentic Systems,\" in ",[609,681,682],{},"Proc. International Conference on Learning Representations (ICLR'25)",[377,684,620],{"href":685,"target":616,"className":686},"https:\u002F\u002Fproceedings.iclr.cc\u002Fpaper_files\u002Fpaper\u002F2025\u002Fhash\u002F36b7acf6f6010652b3f2a433774a66fe-Abstract-Conference.html",[618,619],[604,688,690,691,612,694],{"id":689},"source-8","L. A. Agrawal et al., \"GEPA: Reflective Prompt Evolution Can Outperform Reinforcement Learning,\" in ",[609,692,693],{},"Proc. International Conference on Learning Representations (ICLR'26)",[377,695,620],{"href":696,"target":616,"className":697},"https:\u002F\u002Fproceedings.iclr.cc\u002Fpaper_files\u002Fpaper\u002F2026\u002Fhash\u002F0e9e708b6f48e14fd0ac29e167413f76-Abstract-Conference.html",[618,619],{"title":699,"searchDepth":700,"depth":700,"links":701},"",2,[702,703,704,705,706,710,711],{"id":351,"depth":700,"text":352},{"id":384,"depth":700,"text":385},{"id":420,"depth":700,"text":421},{"id":468,"depth":700,"text":469},{"id":503,"depth":700,"text":504,"children":707},[708],{"id":528,"depth":709,"text":529},3,{"id":566,"depth":700,"text":567},{"id":590,"depth":700,"text":594},"2026-09-27","Reinforcement learning and distillation push a language model in opposite directions, one narrowing its choices and one widening them. A new post-training report from ByteDance argues the two signals work better when they alternate than when their losses are added together, and finds the same separation paying off in credit assignment and in the software wrapped around a frozen model.","md",{"src":360},{"authors":717,"badge":723,"source":725},[718],{"avatar":719,"name":721,"to":722},{"src":720},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":724},"Post-Training",{"name":726,"url":727},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":194,"description":713},"6sx7ettmvNy_PDv6sLHC4uC7Z_sasCHsiSUOU7pltjU",[732,733],null,{"id":734,"title":142,"body":735,"date":927,"description":928,"extension":714,"image":929,"meta":930,"navigation":728,"path":143,"seo":937,"stem":144,"__hash__":938,"_path":143},"insights\u002Fnews\u002Finsights\u002Fcoalitional-alignment-panel-review.md",{"type":332,"value":736,"toc":917},[737,749,755,758,766,770,773,781,789,793,801,824,828,836,844,848,856,864,872],[335,738,339,740,339,744],{"className":739},[338],[341,741,142],{"className":742,"id":743},[344],"a-panel-can-be-safer-than-any-member-of-it",[347,745,748],{"className":746,"id":747},[350],"a-new-theorem-works-out-exactly-when-a-committee-of-misaligned-ai-reviewers-can-still-be-trusted-to-vote-safely","A New Theorem Works Out Exactly When a Committee of Misaligned AI Reviewers Can Still Be Trusted to Vote Safely",[335,750,356,751],{"style":355},[358,752],{"src":753,"alt":754,"style":362},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1687979051782-cd76a9665efb?w=1200&h=750&fit=crop&auto=format","Dozens of mismatched padlocks clipped to a chain link fence, none of them sharing a key, yet together closing off the whole length of it",[364,756,757],{},"A fence covered in mismatched padlocks looks almost silly up close. No two locks share a key, nobody who owns one could open any of the others, and there is no master combination that unlocks the whole row at once. And yet the fence holds. Anyone trying to get through would need to defeat every lock along the stretch they picked, not just the weakest one, and a determined effort that fails at even a single lock still fails. The security here never depended on any individual owner. It depended on how the locks, taken together, happened to cover the whole length of fence.",[364,759,760,761,765],{},"AI systems built to act on a person's behalf run into a version of the same puzzle. An agent that writes code, sends messages, or moves money is usually placed behind some kind of approval step before it can take a consequential action, precisely because it might not be fully aligned with what the user actually wants. Checking every single action with a person defeats the purpose of automating the work in the first place, so a natural next move is to hand that approval job to another AI agent instead. But that just relocates the original worry. The reviewer could be misaligned too, and demanding that it be perfectly aligned is a strange requirement to lean on, given that perfect alignment was the thing nobody trusted the first agent to have. A new paper out of the University of Pennsylvania asks whether a panel of such reviewers, none of them individually aligned with the user, can still deliver a real safety guarantee, and works out precisely when it can ",[374,762,763],{},[377,764,380],{"href":379},".",[347,767,769],{"id":768},"surrounding-the-goal-instead-of-matching-it","Surrounding the Goal Instead of Matching It",[364,771,772],{},"The setup the researchers study is fairly concrete. A user picks some known safe fallback, refusing a tool call, say, or following a cautious default. Whenever an agent proposes doing something else instead, a panel of reviewers looks at that proposal, each reviewer judging it against its own scorecard rather than the user's, and reports a simple approve or disapprove. A rule then counts the votes and decides whether the proposal actually runs.",[364,774,775,776,780],{},"The intuitive fix would be requiring every reviewer to share the user's scorecard. The paper's actual condition is looser and more interesting. What matters is whether the user's scorecard can be built out of the reviewers' scorecards, added together in different amounts, all counted in the same direction, with room left over for something that never actively hurts the user. Call that coverage. The paper proves, exactly, that a voting rule tolerating some number of disapprovals is safe precisely when this coverage condition survives even after that many reviewers are removed from the panel ",[374,777,778],{},[377,779,380],{"href":379},". Coverage can hold even when every single reviewer, taken alone, pulls in a somewhat different direction from the user, which is what makes the condition strictly weaker than asking for an aligned individual anywhere on the panel.",[364,782,783,784,788],{},"That is not just a proof on paper. The researchers ran this coverage check on an existing panel of forty eight reward models, the scoring systems used to judge whether one AI answer is better than another, evaluated against real prompts drawn from a public benchmark. On the large majority of those prompts, the panel's combined scorecards covered the correct answer well enough that a unanimous vote would have blocked every wrong answer, even though no single reward model in the panel was well aligned with correctness on its own ",[374,785,786],{},[377,787,380],{"href":379},". The fence analogy is not just a device for explaining the theorem. It shows up in a panel of models nobody hand picked for alignment.",[347,790,792],{"id":791},"one-step-checked-the-whole-trajectory-covered","One Step Checked, the Whole Trajectory Covered",[364,794,795,796,800],{},"Approving a single action is a useful warm up, but real agents keep acting, watching what happens, and proposing something new based on everything that came before. The paper extends its result into exactly that setting, an agent taking one action after another while a panel reviews each step in turn. The extension is where the result earns its keep. Reviewers at any given step only ever see the immediate choice in front of them, comparing take this action once against fall back to the safe default from here on, they never get to see how the rest of the episode will actually unfold. The paper proves that checking safety this way, one step at a time, is exactly equivalent to safety over the entire run, for any agent that adapts its proposals to the full history it has seen so far ",[374,797,798],{},[377,799,380],{"href":379},". A reviewer never needs to predict the future to guarantee an outcome about it.",[335,802,356,803,356,807,356,811],{"style":519},[358,804],{"src":805,"alt":806,"style":362},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1660784670019-33a36fd82515?w=1200&auto=format&fit=crop","An empty courtroom bench with several vacant seats arranged side by side, built for a group verdict rather than any single voice",[525,808,810],{"style":527,"id":809},"built-for-a-verdict-not-a-vote-count","Built for a Verdict, Not a Vote Count",[364,812,813,814,818,819,823],{"style":532},"A panel of seats like this only means something once the rule for using them is settled, and the paper finds that rule matters more than it might seem. Sincere voting is the obvious strategy when a reviewer only cares about the single decision in front of it, and the paper confirms that a reviewer never does better by lying about its own judgment in a one shot vote ",[374,815,816],{},[377,817,380],{"href":379},". Across a long running series of votes, though, that stops being guaranteed, and reviewers who think ahead can do better for themselves by voting strategically. Demanding unanimous approval turns out to be the rule that keeps every stable outcome safe, so long as the panel's scorecards jointly cover the user's, whatever those individual reviewers privately want. Loosen the rule to tolerate even a single disapproval, and an outcome can become stable, with no reviewer able to improve its own position by changing its vote, while still leaving the user worse off than the fallback, even when every reviewer on that panel is individually well aligned ",[374,820,821],{},[377,822,380],{"href":379},". Tolerance for dissent, not misalignment, is what opens that door.",[347,825,827],{"id":826},"what-numbers-can-do-that-votes-cannot","What Numbers Can Do That Votes Cannot",[364,829,830,831,835],{},"Safety on its own is a cheap guarantee if it comes at the cost of blocking almost everything worth doing. A panel that disapproves of every proposal is perfectly safe and completely useless. The paper calls the other half of this completeness, authorizing enough of what genuinely benefits the user, and finds a real ceiling on what a panel restricted to plain approve or disapprove votes can deliver. When some proposals could help the user and others could hurt, no such rule can guarantee both perfect safety and perfect completeness unless at least one reviewer on the panel is fully aligned to begin with ",[374,832,833],{},[377,834,380],{"href":379},". A single bit per reviewer, it turns out, is not always enough information to work with.",[364,837,838,839,843],{},"The workaround the paper lands on is asking reviewers for a number instead of a bit, how much better or worse the proposal looks relative to the fallback, rather than a simple yes or no. In the same reward model experiments, a committee built by combining the panel's numerical scores authorized the correct answer on well over half of the hardest held out prompts while keeping safety high, comfortably ahead of what the single best individual reviewer managed working alone ",[374,840,841],{},[377,842,380],{"href":379},". Every plain vote based rule tested, by contrast, hit a wall well below that on the same prompts, unable to clear a similarly demanding safety bar without giving up on usefulness almost entirely. The extra bits in a number, rather than any change to who sat on the panel, were what moved the tradeoff.",[347,845,847],{"id":846},"where-the-circularity-actually-goes","Where the Circularity Actually Goes",[364,849,850,851,855],{},"The same research group had already shown something structurally similar in a different setting, where a user facing several competing AI providers, none individually well aligned, could still end up about as well off as with a single aligned model, so long as the user's interests sat somewhere inside the spread of what those providers wanted ",[374,852,853],{},[377,854,394],{"href":393},". That earlier result turns out to be a special case of the coverage condition here, just applied to a market of competing models rather than a panel of reviewers checking one agent's actions.",[364,857,858,859,863],{},"It also sits alongside a different tradition that studies how to control a model that might be actively trying to subvert its own oversight, where a much weaker but trusted model edits or monitors a stronger untrusted one, and holds safety high while giving up comparatively little of what the untrusted model was good at ",[374,860,861],{},[377,862,408],{"href":407},". That work asks how to catch intentional sabotage. This one asks a quieter question, when does a panel's incentives happen to add up in the user's favor regardless of what any individual reviewer intends, and answers it exactly rather than empirically.",[364,865,866,867,871],{},"None of this makes the underlying puzzle disappear. Coverage is not guaranteed just because a panel exists, and the paper's own auditing results suggest that checking whether a real panel actually has it can be a genuinely hard computational problem outside of a few convenient cases. A separate strand of recent work suggests that simply dividing up what a panel is asked to evaluate, so no single review call carries the full weight of a decision, can itself reduce how often oversight quietly fails ",[374,868,869],{},[377,870,438],{"href":437},", a practical habit that fits comfortably next to a theorem about when the panel's math already works out. What the fence and the theorem share is the same reframing. The question worth asking about a review panel may not be whether any single member on it can be trusted, but whether the group, taken together, happens to surround the thing that actually matters.",[335,873,339,875,339,877],{"className":874},[590,591],[347,876,594],{"id":590},[596,878,356,880,356,889,356,898,356,908,339],{"className":879},[599,600,601,602],[604,881,882,883,612,885],{"id":606},"N. Collina et al., \"Delegating Authorization to Misaligned Agents: Coalitional Alignment and Safe Control,\" ",[609,884,611],{},[377,886,620],{"href":887,"target":616,"className":888},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.15803",[618,619],[604,890,891,892,627,894],{"id":623},"N. Collina et al., \"Emergent Alignment via Competition,\" ",[609,893,611],{},[377,895,620],{"href":896,"target":616,"className":897},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.15090",[618,619],[604,899,900,901,903,904],{"id":634},"R. Greenblatt et al., \"AI Control: Improving Safety Despite Intentional Subversion,\" ",[609,902,611],{},", 2023, ",[377,905,620],{"href":906,"target":616,"className":907},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2312.06942",[618,619],[604,909,910,911,612,913],{"id":646},"V. Akinwande et al., \"Sharding Prevents LLM Oversight Failures and Adversarial Exploitation,\" ",[609,912,611],{},[377,914,620],{"href":915,"target":616,"className":916},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.06422",[618,619],{"title":699,"searchDepth":700,"depth":700,"links":918},[919,920,921,924,925,926],{"id":747,"depth":700,"text":748},{"id":768,"depth":700,"text":769},{"id":791,"depth":700,"text":792,"children":922},[923],{"id":809,"depth":709,"text":810},{"id":826,"depth":700,"text":827},{"id":846,"depth":700,"text":847},{"id":590,"depth":700,"text":594},"2026-09-21","When an AI agent is not fully trusted, one fix is to route its proposed actions past a panel of other AI reviewers before anything executes. A new theorem works out exactly when that panel's combined vote can be trusted, even when not one reviewer on it actually shares the user's goals.",{"src":753},{"authors":931,"badge":934,"source":936},[932],{"avatar":933,"name":721,"to":722},{"src":720},{"label":935},"Agent Oversight",{"name":726,"url":727},{"title":142,"description":928},"t4VPdaV54OKZatQC-unabTsh7XwyoOf1UHE5GoIPyXs",1790594464999]