[{"data":1,"prerenderedAt":1024},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival":313,"\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival-surround":612},[4,8,17,21,25,29,33,37,301,305,309],{"title":5,"path":6,"stem":7},"About Thinkata Intelligence","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36},"Leadership","\u002Fleadership","leadership",{"title":38,"path":39,"stem":40,"children":41},"News","\u002Fnews","news",[42,45,69],{"title":43,"path":39,"stem":44},"News & Insights","news\u002Findex",{"title":18,"path":46,"stem":47,"children":48},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[49,53,57,61,65],{"title":50,"path":51,"stem":52},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":54,"path":55,"stem":56},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":58,"path":59,"stem":60},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":62,"path":63,"stem":64},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":66,"path":67,"stem":68},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":70,"stem":71,"children":72},"\u002Fnews\u002Finsights","news\u002Finsights",[73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297],{"title":74,"path":75,"stem":76},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":78,"path":79,"stem":80},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":82,"path":83,"stem":84},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":86,"path":87,"stem":88},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":90,"path":91,"stem":92},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":94,"path":95,"stem":96},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":98,"path":99,"stem":100},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":102,"path":103,"stem":104},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":106,"path":107,"stem":108},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":110,"path":111,"stem":112},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":114,"path":115,"stem":116},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":118,"path":119,"stem":120},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":122,"path":123,"stem":124},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":126,"path":127,"stem":128},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":130,"path":131,"stem":132},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":134,"path":135,"stem":136},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":138,"path":139,"stem":140},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":142,"path":143,"stem":144},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":146,"path":147,"stem":148},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":150,"path":151,"stem":152},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":154,"path":155,"stem":156},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":158,"path":159,"stem":160},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":162,"path":163,"stem":164},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":166,"path":167,"stem":168},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":170,"path":171,"stem":172},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":174,"path":175,"stem":176},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":178,"path":179,"stem":180},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":182,"path":183,"stem":184},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":186,"path":187,"stem":188},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":190,"path":191,"stem":192},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":194,"path":195,"stem":196},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":198,"path":199,"stem":200},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":202,"path":203,"stem":204},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":206,"path":207,"stem":208},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":210,"path":211,"stem":212},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":214,"path":215,"stem":216},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":218,"path":219,"stem":220},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":222,"path":223,"stem":224},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":226,"path":227,"stem":228},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":230,"path":231,"stem":232},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":234,"path":235,"stem":236},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":238,"path":239,"stem":240},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":270,"path":271,"stem":272},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":274,"path":275,"stem":276},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":278,"path":279,"stem":280},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":282,"path":283,"stem":284},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":286,"path":287,"stem":288},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":290,"path":291,"stem":292},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":294,"path":295,"stem":296},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":298,"path":299,"stem":300},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":302,"path":303,"stem":304},"Privacy Policy","\u002Fprivacy","privacy",{"title":306,"path":307,"stem":308},"Research","\u002Fresearch","research",{"title":310,"path":311,"stem":312},"Terms of Service","\u002Fterms","terms",{"id":314,"title":278,"body":315,"date":593,"description":594,"extension":595,"image":596,"meta":597,"navigation":609,"path":279,"seo":610,"stem":280,"__hash__":611},"insights\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival.md",{"type":316,"value":317,"toc":577},"minimark",[318,338,348,361,371,375,383,391,404,421,425,445,458,462,477,481,491,495,498],[319,320,323,324,323,331],"div",{"className":321},[322],"page-title","\n  ",[325,326,330],"h1",{"className":327,"id":329},[328],"page-title__main","when-replay-is-not-an-option","When Replay Is Not an Option",[332,333,337],"h2",{"className":334,"id":336},[335],"page-title__sub","streaming-q-learning-and-sarsa-get-a-second-look","Streaming Q-Learning and SARSA Get a Second Look",[319,339,341,342],{"style":340},"width: 100%; padding: 2%;","\n    ",[343,344],"img",{"src":345,"alt":346,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1772843245076-cb1b1136ceb8?w=1200&auto=format&fit=crop","A fast-flowing forest river passing over rocks without pooling, analogous to streaming reinforcement learning where each experience is used once as it arrives and never stored for later replay","width: 100%; height: auto;",[349,350,351,352,360],"p",{},"A river never pools before it moves on. Water arrives from upstream, passes a given point exactly once, and continues toward the sea without waiting to be collected first. Reinforcement learning, the branch of machine learning where a software agent learns by acting in an environment and receiving rewards, was originally built the same way. Q-learning and SARSA, two classic algorithms that both estimate the future payoff of taking a given action in a given situation, were designed to update from one experience at a time and move on, a mode of operation known as streaming or online learning ",[353,354,355],"sup",{},[356,357,359],"a",{"href":358},"#source-2","[2]",". Both trace back to value iteration, the older dynamic-programming idea of repeatedly correcting a value estimate until it stops changing.",[349,362,363,364,370],{},"Modern deep reinforcement learning rarely runs this way anymore. Since Deep Q-Networks combined Q-learning with neural networks to reach competitive play on Atari games, most systems have leaned on a replay buffer, a large memory bank that stores past experience so it can be resampled again and again in batches ",[353,365,366],{},[356,367,369],{"href":368},"#source-1","[1]",". Averaging many samples together before each update smooths out noise and lets the same experience teach the network more than once. It works well, but it assumes there is somewhere to put the water while it waits its turn.",[332,372,374],{"id":373},"why-the-river-stopped-flowing","Why the River Stopped Flowing",[349,376,377,378,382],{},"Not every system can afford that assumption. A small robot with a modest onboard processor, a wearable sensor, or a system bound by strict data-privacy limits may not have the memory, bandwidth, or permission to keep raw experience around for later replay ",[353,379,380],{},[356,381,359],{"href":358},". When the water cannot be stored, learning has to happen as it passes through, or not happen at all.",[349,384,385,386,390],{},"When researchers tried simply removing the replay buffer from standard deep reinforcement learning algorithms, the results were not encouraging. Learning became unstable or collapsed outright across several benchmark tasks, an effect one recent paper terms the stream barrier ",[353,387,388],{},[356,389,359],{"href":358},". Something about updating incrementally, without the averaging effect of a batch, exposed weaknesses that batch learning had quietly been covering up. Both Q-learning and SARSA compute what is called a temporal difference error, the gap between what the network predicted and what actually happened one step later, then nudge the estimate a small amount toward closing that gap. In a deep network, that nudge is scaled by a step size, and getting the step size wrong turns out to matter far more without a batch to soften the blow.",[349,392,393,394,398,399,403],{},"One proposed fix combines several older ideas that had fallen out of fashion. Eligibility traces, a short-term memory that decays gradually and lets a single reward influence not just the most recent action but several that came before it, were reintroduced alongside layer normalization, a technique that keeps a neural network's internal activity at a stable scale over time, and a sparse pattern of initial connections that reduces interference between unrelated inputs ",[353,395,396],{},[356,397,359],{"href":358},". Bundled into algorithms called stream Q, stream SARSA, and stream actor-critic, these techniques reportedly let a network learn from Atari games, robotic control benchmarks, and a real-world electricity demand forecasting task as each experience arrives, matching or outperforming batch methods on several of them ",[353,400,401],{},[356,402,359],{"href":358},".",[319,405,341,407,341,411,341,417],{"style":406},"width: 100%; margin: 20px 0;",[343,408],{"src":409,"alt":410,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1780034766211-bca35b171cd4?w=1200&auto=format&fit=crop","An industrial valve wheel mounted on a pipe, analogous to choosing a learning step size by feel, where turning too far risks a burst pipe and turning too little changes nothing",[412,413,416],"h3",{"style":414,"id":415},"margin: 1rem 0 0.5rem 0;","tuning-the-flow-without-a-gauge","Tuning the Flow Without a Gauge",[349,418,420],{"style":419},"margin: 0;","A valve wheel like this carries no readout of how much water is actually moving through the pipe behind it. Whoever turns it has to judge the right amount by feel, guided mostly by what happened the last time it was turned this far. Step size, the setting that controls how large a correction a learning algorithm makes after each mistake, has traditionally been chosen the same way in streaming reinforcement learning. Too far, and the update destabilizes the model. Too little, and nothing changes fast enough to matter.",[332,422,424],{"id":423},"choosing-the-step-size-on-purpose","Choosing the Step Size on Purpose",[349,426,427,428,434,435,439,440,444],{},"A separate line of work asks a different question about that same wheel. Rather than picking a step size measured in the units of a network's internal weights, why not decide first how much the model's actual prediction should change, then solve backward for whichever step size makes that happen, sidestepping the guesswork that made the stream barrier so damaging in the first place ",[353,429,430],{},[356,431,433],{"href":432},"#source-3","[3]",". Researchers working alongside Richard Sutton, one of the field's founding figures, call the result intentional updates. Instead of fixing how far to turn the wheel, the method fixes how much water should flow and calculates the turn needed to get there ",[353,436,437],{},[356,438,433],{"href":432},". Applied to temporal difference learning and to Q-learning, the approach is reported to reach streaming performance that is often comparable to batch and replay-buffer methods across the domains tested so far ",[353,441,442],{},[356,443,433],{"href":432},". Whether that holds up on tasks with much longer horizons than the current benchmarks is a question the authors leave open, and one further work will need to settle.",[319,446,341,447,341,451,341,455],{"style":406},[343,448],{"src":449,"alt":450,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1611663806011-b37e091090f0?w=1200&auto=format&fit=crop","A small green microcontroller development board with visible chips and pins, analogous to a resource-constrained device that has room to hold only the current moment of experience, not a warehouse of stored samples",[412,452,454],{"style":414,"id":453},"learning-on-the-device-itself","Learning on the Device Itself",[349,456,457],{"style":419},"A board this size holds only what is happening right now, with no spare memory set aside for a warehouse of stored samples. That is not an edge case for streaming reinforcement learning so much as close to the point of the exercise.",[332,459,461],{"id":460},"from-simulation-to-real-hardware","From Simulation to Real Hardware",[349,463,464,465,471,472,476],{},"Researchers recently tested an incremental policy gradient method, a family of algorithms that adjust a decision-making policy directly rather than through an intermediate value estimate, under the name Action Value Gradient. On simulated robotic control benchmarks, it was reportedly the only incremental method among those compared that learned effectively without a replay buffer, target network, or batch update ",[353,466,467],{},[356,468,470],{"href":469},"#source-4","[4]",". It was later used to train a robotic manipulator arm and a mobile robot with only real-time, incremental updates, which the authors describe as the first demonstration of effective deep reinforcement learning on physical robots restricted to incremental learning ",[353,473,474],{},[356,475,470],{"href":469},". The result matters less because of the specific robots involved and more because it suggests the resource savings from dropping batch machinery do not necessarily come at the cost of learning ability, at least on the hardware and tasks tested.",[332,478,480],{"id":479},"where-the-simpler-algorithms-already-work","Where the Simpler Algorithms Already Work",[349,482,483,484,490],{},"None of this implies classical Q-learning and SARSA need deep networks or streaming tricks to be useful today. In a 2025 comparison of microgrid energy management, a setting involving batteries, solar generation, and shifting demand, researchers found that plain tabular Q-learning and SARSA still produced meaningful cost savings compared with having no learning agent at all, even though a deep Q-network outperformed both by a further margin ",[353,485,486],{},[356,487,489],{"href":488},"#source-5","[5]",". Whether that gap between simple and deep methods narrows or widens as more real energy systems get instrumented is a question best left to people closer to that domain. From a systems perspective, though, it is a useful reminder that the current streaming revival is not solving a problem that never existed before. It is trying to bring the modeling capacity of deep networks into settings where the older, simpler algorithms already had to operate without a replay buffer by design.",[332,492,494],{"id":493},"what-this-suggests","What This Suggests",[349,496,497],{},"Streaming reinforcement learning was not so much invented recently as recovered. Q-learning and SARSA were streaming algorithms from the outset, descendants of value iteration's habit of repeatedly correcting an estimate against a moving target. Batch learning became dominant afterward, once neural networks entered the picture, partly because batching hid instabilities that incremental updates expose immediately. The recent work on the stream barrier, intentional step sizes, and incremental policy gradients does not eliminate that instability. It builds tools aimed at the constraint that made replay buffers attractive in the first place, limited onboard hardware and experience that can only be used once. Whether these methods extend cleanly to longer, messier tasks than the current benchmarks cover remains an open question, one likely to be answered one small robot and one edge device at a time.",[319,499,323,503,323,506],{"className":500},[501,502],"references","mt-8",[332,504,505],{"id":501},"References",[507,508,341,514,341,532,341,544,341,555,341,565,323],"ol",{"className":509},[510,511,512,513],"list-decimal","list-inside","space-y-2","mt-4",[515,516,518,519,523,524],"li",{"id":517},"source-1","V. Mnih et al., \"Human-level control through deep reinforcement learning,\" ",[520,521,522],"em",{},"Nature",", vol. 518, no. 7540, pp. 529–533, 2015. DOI: ",[356,525,531],{"href":526,"target":527,"className":528},"https:\u002F\u002Fdoi.org\u002F10.1038\u002Fnature14236","_blank",[529,530],"text-blue-600","underline","[Online]",[515,533,535,536,539,540],{"id":534},"source-2","M. Elsayed et al., \"Streaming Deep Reinforcement Learning Finally Works,\" ",[520,537,538],{},"arXiv",", 2024, ",[356,541,531],{"href":542,"target":527,"className":543},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2410.14606",[529,530],[515,545,547,548,550,551],{"id":546},"source-3","A. Sharifnassab et al., \"Intentional Updates for Streaming Reinforcement Learning,\" ",[520,549,538],{},", 2026, ",[356,552,531],{"href":553,"target":527,"className":554},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.19033",[529,530],[515,556,558,559,539,561],{"id":557},"source-4","G. Vasan et al., \"Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers,\" ",[520,560,538],{},[356,562,531],{"href":563,"target":527,"className":564},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2411.15370",[529,530],[515,566,568,569,572,573],{"id":567},"source-5","S. Ramesh et al., \"Comparative analysis of Q-learning, SARSA, and deep Q-network for microgrid energy management,\" ",[520,570,571],{},"Scientific Reports",", vol. 15, article 694, 2025. DOI: ",[356,574,531],{"href":575,"target":527,"className":576},"https:\u002F\u002Fdoi.org\u002F10.1038\u002Fs41598-024-83625-8",[529,530],{"title":578,"searchDepth":579,"depth":579,"links":580},"",2,[581,582,586,589,590,591,592],{"id":336,"depth":579,"text":337},{"id":373,"depth":579,"text":374,"children":583},[584],{"id":415,"depth":585,"text":416},3,{"id":423,"depth":579,"text":424,"children":587},[588],{"id":453,"depth":585,"text":454},{"id":460,"depth":579,"text":461},{"id":479,"depth":579,"text":480},{"id":493,"depth":579,"text":494},{"id":501,"depth":579,"text":505},"2026-07-19","New research revisits streaming Q-learning and SARSA, the original one-sample-at-a-time reinforcement learning algorithms, examining why deep versions became unstable without a replay buffer and what recent step-size and eligibility trace fixes suggest for on-device learning.","md",{"src":345},{"authors":598,"badge":604,"source":606},[599],{"avatar":600,"name":602,"to":603},{"src":601},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":605},"Reinforcement Learning",{"name":607,"url":608},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":278,"description":594},"x71f018zPUk62pRKT3W7CaR_Z0cFN4ccOwkGaz2NRDg",[613,832],{"id":614,"title":74,"body":615,"date":820,"description":821,"extension":595,"image":822,"meta":823,"navigation":609,"path":75,"seo":830,"stem":76,"__hash__":831,"_path":75},"insights\u002Fnews\u002Finsights\u002Fadaptive-context-memory.md",{"type":316,"value":616,"toc":808},[617,629,635,638,645,649,657,664,677,685,689,697,705,718,722,735,747,749,752],[319,618,323,620,323,624],{"className":619},[322],[325,621,74],{"className":622,"id":623},[328],"nothing-gets-deleted-just-blurred",[332,625,628],{"className":626,"id":627},[335],"long-context-models-are-learning-when-to-zoom-back-in","Long-Context Models Are Learning When to Zoom Back In",[319,630,341,631],{"style":340},[343,632],{"src":633,"alt":634,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1623475329493-889804e377f8?w=1200&auto=format&fit=crop","A curling strip of 35mm slide film with visible frames of buildings and landscapes across its length, analogous to how long-context inference now keeps a compressed thumbnail of most of a conversation rather than deleting it outright",[349,636,637],{},"A strip of developed film holds every frame a photographer shot, but only as a thumbnail. Nothing on the strip is discarded, and nothing on it is sharp enough to read without a loupe or an enlarger. The decision about which frame gets printed at full size happens later, once someone knows which shot actually matters. Large language models handling long documents or long conversations face a version of the same choice, and for the last few years, most systems have chosen the opposite path, throwing frames away entirely rather than shrinking them.",[349,639,640,641,403],{},"The technical name for the film strip inside a language model is the key-value cache, commonly shortened to KV cache. Every time a transformer, the neural network architecture behind most modern language models, processes a token, it computes a key and a value vector for that token in every attention layer, the mechanism that lets the model weigh how much each earlier token matters to the one being generated now. Those vectors get stored so future tokens can attend back to them without recomputing them from scratch. Research from Google on scaling transformer inference showed early on that this cache grows linearly with how much text the model has read, and that the resulting memory pressure, not raw compute, is often what limits how long a context a deployed model can actually afford to hold ",[353,642,643],{},[356,644,369],{"href":368},[332,646,648],{"id":647},"deciding-once-and-living-with-it","Deciding Once and Living With It",[349,650,651,652,656],{},"Most existing fixes to this problem make an early, permanent call. Token eviction methods watch which tokens receive the least attention early on and drop them from the cache for good. Semantic compression methods group tokens into chunks during the initial read and replace each chunk with a single averaged summary before generation even starts. Both approaches work, and both share the same limitation, according to a team from the University of British Columbia and Microsoft Research studying long-context KV caching. A decision about what to keep gets made once, at the beginning, before the model has any idea which later question will actually depend on which earlier passage ",[353,653,654],{},[356,655,359],{"href":358},". Evidence that looked irrelevant in the first paragraph of a legal filing or a codebase might turn out to be the exact clause a later question hinges on, and once it has been evicted or blended into an average, there is no way back.",[349,658,659,660,403],{},"The researchers propose something they call SeKV, short for semantic KV cache, which keeps every span of text in two forms at once rather than picking one. A lightweight summary vector for each span stays resident on the GPU, cheap to scan and used only to decide whether that span looks relevant to the current decoding step. A separate, more detailed representation of the same span, built from a technique called singular value decomposition that factors a block of numbers into a smaller set of components capturing most of its structure, sits on the CPU instead, waiting to be fetched only if the summary suggests it is worth the trip. Segment boundaries themselves come from a signal already available for free during the model's first pass over the text, a measure of how surprising each token is given what came before it, which tends to spike at genuine topic shifts and drop within a coherent run of text ",[353,661,662],{},[356,663,359],{"href":358},[319,665,341,666,341,670,341,674],{"style":406},[343,667],{"src":668,"alt":669,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1722080767146-0ae3a8ff46a0?w=1200&auto=format&fit=crop","A satellite view of a dark lake surrounded by mountains, showing broad shapes clearly but withholding finer ground-level detail until the view is zoomed in, analogous to a coarse memory summary that a model can request higher resolution from on demand",[412,671,673],{"style":414,"id":672},"a-coarse-view-that-can-be-zoomed","A Coarse View That Can Be Zoomed",[349,675,676],{"style":419},"A satellite photograph like this one shows enough to locate a shoreline or a mountain range, but not enough to make out a building or a road. Getting that level of detail means requesting a closer pass over one particular area, not over the whole frame. SeKV applies the same logic to context, routing over cheap summaries first and only reconstructing full detail for the handful of spans a given decoding step actually needs.",[349,678,679,680,684],{},"Whether a given span deserves that closer look gets decided by a small trained component, referred to as a zoom-in mechanism, that scores each summary against the model's current query and expands only the spans that clear a learned threshold. Across four long-context benchmarks, the reported result was a 5.9 percent average improvement over the strongest semantic-compression baseline tested, alongside a 53.3 percent reduction in GPU memory compared with keeping the full cache at a 128,000-token context length, achieved while adding fewer than 0.05 percent additional trainable parameters to a frozen base model ",[353,681,682],{},[356,683,359],{"href":358},". The efficiency gain and the accuracy gain showing up together is worth noting mainly because compression methods usually have to trade one for the other.",[332,686,688],{"id":687},"a-token-that-asks-for-a-refresh","A Token That Asks for a Refresh",[349,690,691,692,696],{},"A separate team, working across several Chinese research labs, tackled a closely related question from another angle. Rather than deciding what to keep once and reconstructing detail on demand, their system, called PReM for preserve and refresh memory, lets the model itself decide when its current compressed view of the context has gone stale and needs to be rebuilt ",[353,693,694],{},[356,695,433],{"href":432},". PReM trains a special memory token that the model can emit mid-generation. Emitting it triggers a fresh look back over the full context, re-selecting which chunks deserve to be kept at full token-level detail and which get replaced with an averaged stand-in, based on whatever the model is trying to figure out at that specific step rather than on whatever seemed important when the context was first read.",[349,698,699,700,704],{},"Training a model to condition its own generation on a memory state that changes partway through, rather than staying fixed for the whole response, required the researchers to split each generation step into two separate forward passes, one that handles the refresh decision and one that generates conditioned on the result. On a set of question-answering benchmarks using 32,000-token contexts, PReM reportedly outperformed eight established KV-cache and context-compression baselines under both 16 times and 32 times compression, improving average exact-match and F1 scores by up to 10.23 and 12.55 points respectively over the strongest baseline tested ",[353,701,702],{},[356,703,433],{"href":432},". A smaller 3-billion-parameter version of the model was also reported to outperform some larger 7-billion-parameter compression baselines, which the authors take as a sign that learning when to refresh can partly substitute for raw model scale, at least on the benchmarks tried so far.",[319,706,341,707,341,711,341,715],{"style":406},[343,708],{"src":709,"alt":710,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1573166364266-356ef04ae798?w=1200&auto=format&fit=crop","A person writing on a dry-erase whiteboard, part of the surface freshly wiped clean while other notes remain, analogous to a memory token that triggers a partial refresh of what a model keeps in view without discarding the underlying source material",[412,712,714],{"style":414,"id":713},"refreshing-the-board-without-losing-the-room","Refreshing the Board Without Losing the Room",[349,716,717],{"style":419},"Wiping part of a whiteboard clean does not erase the meeting that produced the notes, only the summary written down at the time. PReM's memory token works on a similar principle, refreshing the compressed view the model is currently attending to while the full source context stays available underneath, ready to be resummarized differently the next time a refresh is triggered.",[332,719,721],{"id":720},"memory-that-outlives-a-single-document","Memory That Outlives a Single Document",[349,723,724,725,729,730,734],{},"Both SeKV and PReM operate within a single long document or a single extended answer. A different line of work asks what changes when the relevant memory spans many separate conversations with the same user over time, a setting closer to how a deployed assistant actually gets used. A team centered at the University of Electronic Science and Technology of China proposes LightMem, which splits an agent's memory into three tiers, a short-term store for the immediate conversation, a mid-term store of reusable summaries from recent sessions, and a long-term store of consolidated knowledge, each managed by a small, purpose-built language model rather than a large one ",[353,726,727],{},[356,728,470],{"href":469},". Retrieval happens in two cheap stages, a coarse vector search followed by a semantic consistency check, before anything gets handed to the main model. Reported results on a long-term dialogue benchmark showed roughly a 2.5-point average gain in F1 score across model scales, with a median end-to-end latency around 581 milliseconds and retrieval alone completing in about 83 milliseconds ",[353,731,732],{},[356,733,470],{"href":469},". The appeal of the approach, from a systems perspective, sits less in the accuracy number and more in the latency figure, since memory operations that run on small models rather than the main large model avoid competing for the same expensive compute.",[349,736,737,738,742,743,403],{},"A related but distinct idea comes from a team at the University of Science and Technology of China, who point out that agents which compress a long document into memory through a single linear pass, reading start to finish and updating a running summary as they go, tend to prune evidence early that only turns out to matter once a much later part of the document has been read ",[353,739,740],{},[356,741,489],{"href":488},". Their system, ReMemR1, lets an agent issue what the authors call a callback query mid-reasoning, retrieving an earlier memory state instead of only ever moving forward through it, and trains the behavior with a reward signal that combines the final answer's correctness with denser, step-level feedback about whether a given callback was actually useful. On long-context question-answering benchmarks, the reported result was upward of a 20 percent relative reduction in error rate compared with linear memory baselines, with the retrieval mechanism adding less than 0.2 percent to overall computation time ",[353,744,745],{},[356,746,489],{"href":488},[332,748,494],{"id":493},[349,750,751],{},"Four independent groups converging on some version of the same idea, that compression decisions should stay reversible and resolution should be allocated on demand rather than fixed up front, is a pattern worth watching rather than a settled conclusion. Every result above comes from a specific benchmark and a specific context length, 32,000 tokens for PReM, 128,000 for SeKV, documents padded to a few thousand entries for ReMemR1, and none of them yet speaks directly to million-token agent contexts or to workloads where the CPU-to-GPU transfer that SeKV and similar systems depend on becomes a bottleneck of its own under real production load. Whether the accuracy gains reported here persist once these methods leave curated benchmarks and meet the messier, longer-horizon contexts that production agents actually accumulate remains an open question, and one that will likely need answering system by system rather than in the abstract.",[319,753,323,755,323,757],{"className":754},[501,502],[332,756,505],{"id":501},[507,758,341,760,341,771,341,780,341,789,341,798,323],{"className":759},[510,511,512,513],[515,761,762,763,766,767],{"id":517},"R. Pope et al., \"Efficiently Scaling Transformer Inference,\" ",[520,764,765],{},"Proceedings of the Sixth Conference on Machine Learning and Systems (MLSys 2023)",", 2023. DOI: ",[356,768,531],{"href":769,"target":527,"className":770},"https:\u002F\u002Fdoi.org\u002F10.48550\u002FarXiv.2211.05102",[529,530],[515,772,773,774,550,776],{"id":534},"A. Abaskohi et al., \"SeKV: Resolution-Adaptive KV Cache with Hierarchical Semantic Memory for Long-Context LLM Inference,\" ",[520,775,538],{},[356,777,531],{"href":778,"target":527,"className":779},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.31145",[529,530],[515,781,782,783,550,785],{"id":546},"B. Yu et al., \"PReM: Learning What to Preserve and When to Refresh for Context Compression,\" ",[520,784,538],{},[356,786,531],{"href":787,"target":527,"className":788},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2607.14327",[529,530],[515,790,791,792,550,794],{"id":557},"J. Zhang et al., \"Lightweight LLM Agent Memory with Small Language Models,\" ",[520,793,538],{},[356,795,531],{"href":796,"target":527,"className":797},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.07798",[529,530],[515,799,800,801,803,804],{"id":567},"Y. Shi et al., \"Look Back to Reason Forward: Revisitable Memory for Long-Context LLM Agents,\" ",[520,802,538],{},", 2025, ",[356,805,531],{"href":806,"target":527,"className":807},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.23040",[529,530],{"title":578,"searchDepth":579,"depth":579,"links":809},[810,811,814,817,818,819],{"id":627,"depth":579,"text":628},{"id":647,"depth":579,"text":648,"children":812},[813],{"id":672,"depth":585,"text":673},{"id":687,"depth":579,"text":688,"children":815},[816],{"id":713,"depth":585,"text":714},{"id":720,"depth":579,"text":721},{"id":493,"depth":579,"text":494},{"id":501,"depth":579,"text":505},"2026-07-27","Long-context language models have relied on discarding or freezing most of what they read to keep memory costs down. New research on resolution-adaptive caching and refreshable memory tokens suggests a different approach, keeping everything at low resolution and reconstructing detail only when a later step actually needs it.",{"src":633},{"authors":824,"badge":827,"source":829},[825],{"avatar":826,"name":602,"to":603},{"src":601},{"label":828},"Long-Context Inference",{"name":607,"url":608},{"title":74,"description":821},"JZHObIsLZXHYqMm_rMP1pKlqCXS6qzRMaqUmKazEsmQ",{"id":833,"title":214,"body":834,"date":1012,"description":1013,"extension":595,"image":1014,"meta":1015,"navigation":609,"path":215,"seo":1022,"stem":216,"__hash__":1023,"_path":215},"insights\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality.md",{"type":316,"value":835,"toc":1005},[836,848,854,857,870,874,880,888,896,900,906,914,922,926,934,945],[319,837,323,839,323,843],{"className":838},[322],[325,840,214],{"className":841,"id":842},[328],"the-expensive-default",[332,844,847],{"className":845,"id":846},[335],"llm-routing-promises-to-send-easy-questions-to-cheap-models-and-research-keeps-finding-it-reaches-for-the-expensive-one-instead","LLM routing promises to send easy questions to cheap models, and research keeps finding it reaches for the expensive one instead",[319,849,341,850],{"style":340},[343,851],{"src":852,"alt":853,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1744745439306-f16d00620b38?w=1200&auto=format&fit=crop","A cluster of directional signs on a single post pointing toward many different destinations, standing in for the decision a production AI system faces on every incoming request, which of several available models should handle this one",[349,855,856],{},"Every product built on a large language model eventually runs into a question that has nothing to do with prompts or fine-tuning. A request arrives, and something has to decide which model answers it. Send everything to the most capable model available and the bill grows in proportion to traffic, whether or not the traffic needed that much capability. Send everything to a cheaper model and some fraction of users get worse answers than the product could have given them.",[349,858,859,860,864,865,869],{},"The engineering answer to that tradeoff is a router, a lightweight system placed in front of the language models that reads each incoming request and assigns it to one of them. Early academic work on this problem found that many questions do not need the strongest available model at all. A quality-aware router trained to send only the harder queries to a large model reduced calls to that model by up to 40 percent with no measurable drop in response quality ",[353,861,862],{},[356,863,369],{"href":368},". A follow-up framework refined the idea using human preference data collected from model comparisons, and on one benchmark it matched 95 percent of a strong model's score while sending only 13.4 percent of requests to that model ",[353,866,867],{},[356,868,359],{"href":358},". The premise looked sound. Most requests are easier than the hardest request a product will ever see, and a router that can tell the difference should save money without anyone noticing.",[332,871,873],{"id":872},"reaching-for-the-expensive-model-anyway","Reaching for the Expensive Model Anyway",[319,875,341,876],{"style":340},[343,877],{"src":878,"alt":879,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1740560516658-5a94b0b715ed?w=1200&auto=format&fit=crop","A mercury thermometer planted in sand against a blue sky, its red column climbing toward the top of the scale, standing in for a routing system that keeps escalating toward its most expensive, most capable model even when a lower reading would perform just as well",[349,881,882,883,887],{},"Deployed routers turn out to behave differently from the ones described in the papers that introduced the idea. A 2026 study tracking router behavior as the allowed budget per query increases found something researchers had not previously isolated as a distinct failure. Instead of spreading requests across the available models based on difficulty, the call rate of the single most expensive model climbs steadily and eventually saturates near 100 percent, the reading pushed toward the top of the scale regardless of whether the query in front of it needed that much, even though a hindsight-optimal router on the same benchmark uses that model for fewer than 20 percent of queries under the same budget ",[353,884,885],{},[356,886,433],{"href":432},". The researchers behind that finding named it routing collapse, and traced the cause to something structural rather than a training bug. Most queries have several models bunched close together in quality, and when the top two or three candidates are nearly tied, a routing model's small prediction errors are enough to flip which one looks best. As the budget grows and more models become affordable, that instability consistently pulls the decision toward the strongest, most expensive option, because a router comparing near-ties has no reliable way to settle on the cheaper one instead.",[349,889,890,891,895],{},"The opposite failure shows up just as often. A benchmark spanning more than 400,000 queries across 21 datasets and 33 models found that current routers, even ones sold as commercial products, still fail to land on the one correct model for a meaningful slice of requests. On the subset of test queries where only one to three of the available models could answer correctly, two widely used routing methods hit just 23 to 25 percent accuracy at identifying which one ",[353,892,893],{},[356,894,470],{"href":469},". So a router can spend too much by defaulting to the strongest model when it did not need to, and separately miss the model that actually had the right answer when specificity mattered. Both problems come from the same root cause. Comparing models that perform almost identically on most requests is a harder discrimination task than comparing models with a wide quality gap, and most real traffic falls into the first category.",[332,897,899],{"id":898},"naming-the-target-instead-of-guessing","Naming the Target Instead of Guessing",[319,901,341,902],{"style":340},[343,903],{"src":904,"alt":905,"style":347},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1629122433131-53e850a3a2ce?w=1200&auto=format&fit=crop","A round target with several arrows clustered around, but not on, the center ring, standing in for the scatter that results when a routing system is tuned through indirect proxies instead of being told exactly where the target sits",[349,907,908,909,913],{},"Most routers ask an operator to set an indirect parameter, a cost threshold or a confidence cutoff, and observe afterward what accuracy that setting happened to produce. The result lands somewhere near the outcome the operator wanted, rarely exactly on it. One router built specifically against this complaint takes the opposite approach, accepting a target accuracy as a direct input rather than something inferred from unrelated knobs ",[353,910,911],{},[356,912,489],{"href":488},". Internally it tracks how far recent decisions have drifted from the requested target and adjusts its own aggressiveness in real time, letting a single trained system serve a whole range of accuracy requirements without retraining for each one.",[349,915,916,917,921],{},"The reported results suggest the direct-target approach closes much of the gap described above. Tested against a baseline that also tried to enforce a minimum accuracy level, the target-based router met its stated floor consistently, where the baseline met it only 22 percent of the time. On one benchmark it reached within 1.3 percent of the best achievable accuracy while cutting cost by as much as 89.8 percent compared to always using the strongest model ",[353,918,919],{},[356,920,489],{"href":488},". The gain is not that the underlying models changed. It is that the system was finally asked to hold a specific, checkable commitment instead of a proxy for one.",[332,923,925],{"id":924},"where-the-discipline-has-to-live","Where the Discipline Has to Live",[349,927,928,929,933],{},"Cost and quality are not the only axis worth tracking, and treating them as the whole problem hides a third variable that shows up the moment routing decisions reach production. Two models can score almost identically on both accuracy and price and still differ enormously in how long a response takes to arrive. One documented pair produced comparable results at comparable cost with response times of 32 seconds and 262 seconds ",[353,930,931],{},[356,932,470],{"href":469},". A router optimizing on two dimensions has no way to notice that gap, and a user waiting eight times longer for an equivalent answer will notice it regardless of what the router's dashboard says about savings.",[349,935,936,937,941,942,944],{},"None of this argues against routing. It argues against treating a cost threshold as a stand-in for the outcome a product actually promises its users. A routing layer is only as trustworthy as the evaluation underneath it, the same argument that applies to any system where one model's output stands in for a judgment about quality ",[353,938,939],{},[356,940,359],{"href":358},", a point explored further in ",[356,943,206],{"href":207},". A platform doing model routing should treat the accuracy floor as the product commitment and the routing table as the implementation detail underneath it, not the other way around. The question worth asking before shipping a router is not how much it saves on average. It is what happens on the request where the cheap model was wrong and nobody was checking.",[319,946,323,948,323,950],{"className":947},[501,502],[332,949,505],{"id":501},[507,951,341,953,341,964,341,975,341,985,341,996,323],{"className":952},[510,511,512,513],[515,954,955,956,959,960],{"id":517},"D. Ding, A. Mallick, C. Wang, R. Sim, S. Mukherjee, V. Rühle, L. V. S. Lakshmanan, and A. H. Awadallah, \"Hybrid LLM: Cost-Efficient and Quality-Aware Query Routing,\" in ",[520,957,958],{},"Proc. Twelfth International Conference on Learning Representations (ICLR 2024)",", 2024. ",[356,961,531],{"href":962,"target":527,"className":963},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2404.14618",[529,530],[515,965,966,967,970,971],{"id":534},"I. Ong, A. Almahairi, V. Wu, W. Chiang, T. Wu, J. E. Gonzalez, M. W. Kadous, and I. Stoica, \"RouteLLM: Learning to Route LLMs with Preference Data,\" in ",[520,968,969],{},"Proc. International Conference on Learning Representations (ICLR 2025)",", 2025. ",[356,972,531],{"href":973,"target":527,"className":974},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2406.18665",[529,530],[515,976,977,978,980,981],{"id":546},"G. Lai and H.-J. Ye, \"When Routing Collapses: On the Degenerate Convergence of LLM Routers,\" ",[520,979,538],{},", 2026. ",[356,982,531],{"href":983,"target":527,"className":984},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2602.03478",[529,530],[515,986,987,988,991,992],{"id":557},"H. Li, Y. Zhang, Z. Guo, C. Wang, S. Tang, Q. Zhang, Y. Chen, B. Qi, P. Ye, L. Bai, Z. Wang, and S. Hu, \"LLMRouterBench: A Massive Benchmark and Unified Framework for LLM Routing,\" in ",[520,989,990],{},"Findings of the Association for Computational Linguistics: ACL 2026",", 2026, pp. 37733–37754. ",[356,993,531],{"href":994,"target":527,"className":995},"https:\u002F\u002Faclanthology.org\u002F2026.findings-acl.1881\u002F",[529,530],[515,997,998,999,980,1001],{"id":567},"A. S. Bhatti, V. Vaddina, and D. Birru, \"PROTEUS: SLA-Aware Routing via Lagrangian RL for Multi-LLM Serving Systems,\" ",[520,1000,538],{},[356,1002,531],{"href":1003,"target":527,"className":1004},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.19402",[529,530],{"title":578,"searchDepth":579,"depth":579,"links":1006},[1007,1008,1009,1010,1011],{"id":846,"depth":579,"text":847},{"id":872,"depth":579,"text":873},{"id":898,"depth":579,"text":899},{"id":924,"depth":579,"text":925},{"id":501,"depth":579,"text":505},"2026-07-14","LLM routing promises to send easy questions to cheap models and hard questions to expensive ones. Recent research keeps finding that the systems built to do this default to the expensive model anyway, and still miss the right pick when it counts.",{"src":852},{"authors":1016,"badge":1019,"source":1021},[1017],{"avatar":1018,"name":602,"to":608},{"src":601},{"label":1020},"AI Infrastructure",{"name":607,"url":608},{"title":214,"description":1013},"ThnI8PTU6GLgSiv9QGVXFfqAtl-64CfR7BfnMA4ryxE",1787055018842]