[{"data":1,"prerenderedAt":1037},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Fadaptive-context-memory":313,"\u002Fnews\u002Finsights\u002Fadaptive-context-memory-surround":595},[4,8,17,21,25,29,33,37,301,305,309],{"title":5,"path":6,"stem":7},"About Thinkata Intelligence","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36},"Leadership","\u002Fleadership","leadership",{"title":38,"path":39,"stem":40,"children":41},"News","\u002Fnews","news",[42,45,69],{"title":43,"path":39,"stem":44},"News & Insights","news\u002Findex",{"title":18,"path":46,"stem":47,"children":48},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[49,53,57,61,65],{"title":50,"path":51,"stem":52},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":54,"path":55,"stem":56},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":58,"path":59,"stem":60},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":62,"path":63,"stem":64},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":66,"path":67,"stem":68},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":70,"stem":71,"children":72},"\u002Fnews\u002Finsights","news\u002Finsights",[73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297],{"title":74,"path":75,"stem":76},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":78,"path":79,"stem":80},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":82,"path":83,"stem":84},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":86,"path":87,"stem":88},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":90,"path":91,"stem":92},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":94,"path":95,"stem":96},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":98,"path":99,"stem":100},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":102,"path":103,"stem":104},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":106,"path":107,"stem":108},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":110,"path":111,"stem":112},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":114,"path":115,"stem":116},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":118,"path":119,"stem":120},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":122,"path":123,"stem":124},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":126,"path":127,"stem":128},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":130,"path":131,"stem":132},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":134,"path":135,"stem":136},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":138,"path":139,"stem":140},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":142,"path":143,"stem":144},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":146,"path":147,"stem":148},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":150,"path":151,"stem":152},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":154,"path":155,"stem":156},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":158,"path":159,"stem":160},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":162,"path":163,"stem":164},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":166,"path":167,"stem":168},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":170,"path":171,"stem":172},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":174,"path":175,"stem":176},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":178,"path":179,"stem":180},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":182,"path":183,"stem":184},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":186,"path":187,"stem":188},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":190,"path":191,"stem":192},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":194,"path":195,"stem":196},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":198,"path":199,"stem":200},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":202,"path":203,"stem":204},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":206,"path":207,"stem":208},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":210,"path":211,"stem":212},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":214,"path":215,"stem":216},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":218,"path":219,"stem":220},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":222,"path":223,"stem":224},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":226,"path":227,"stem":228},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":230,"path":231,"stem":232},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":234,"path":235,"stem":236},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":238,"path":239,"stem":240},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":270,"path":271,"stem":272},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":274,"path":275,"stem":276},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":278,"path":279,"stem":280},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":282,"path":283,"stem":284},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":286,"path":287,"stem":288},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":290,"path":291,"stem":292},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":294,"path":295,"stem":296},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":298,"path":299,"stem":300},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":302,"path":303,"stem":304},"Privacy Policy","\u002Fprivacy","privacy",{"title":306,"path":307,"stem":308},"Research","\u002Fresearch","research",{"title":310,"path":311,"stem":312},"Terms of Service","\u002Fterms","terms",{"id":314,"title":74,"body":315,"date":576,"description":577,"extension":578,"image":579,"meta":580,"navigation":592,"path":75,"seo":593,"stem":76,"__hash__":594},"insights\u002Fnews\u002Finsights\u002Fadaptive-context-memory.md",{"type":316,"value":317,"toc":561},"minimark",[318,337,347,351,363,367,377,384,401,409,413,423,431,444,448,463,477,481,484],[319,320,323,324,323,330],"div",{"className":321},[322],"page-title","\n  ",[325,326,74],"h1",{"className":327,"id":329},[328],"page-title__main","nothing-gets-deleted-just-blurred",[331,332,336],"h2",{"className":333,"id":335},[334],"page-title__sub","long-context-models-are-learning-when-to-zoom-back-in","Long-Context Models Are Learning When to Zoom Back In",[319,338,340,341],{"style":339},"width: 100%; padding: 2%;","\n    ",[342,343],"img",{"src":344,"alt":345,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1623475329493-889804e377f8?w=1200&auto=format&fit=crop","A curling strip of 35mm slide film with visible frames of buildings and landscapes across its length, analogous to how long-context inference now keeps a compressed thumbnail of most of a conversation rather than deleting it outright","width: 100%; height: auto;",[348,349,350],"p",{},"A strip of developed film holds every frame a photographer shot, but only as a thumbnail. Nothing on the strip is discarded, and nothing on it is sharp enough to read without a loupe or an enlarger. The decision about which frame gets printed at full size happens later, once someone knows which shot actually matters. Large language models handling long documents or long conversations face a version of the same choice, and for the last few years, most systems have chosen the opposite path, throwing frames away entirely rather than shrinking them.",[348,352,353,354,362],{},"The technical name for the film strip inside a language model is the key-value cache, commonly shortened to KV cache. Every time a transformer, the neural network architecture behind most modern language models, processes a token, it computes a key and a value vector for that token in every attention layer, the mechanism that lets the model weigh how much each earlier token matters to the one being generated now. Those vectors get stored so future tokens can attend back to them without recomputing them from scratch. Research from Google on scaling transformer inference showed early on that this cache grows linearly with how much text the model has read, and that the resulting memory pressure, not raw compute, is often what limits how long a context a deployed model can actually afford to hold ",[355,356,357],"sup",{},[358,359,361],"a",{"href":360},"#source-1","[1]",".",[331,364,366],{"id":365},"deciding-once-and-living-with-it","Deciding Once and Living With It",[348,368,369,370,376],{},"Most existing fixes to this problem make an early, permanent call. Token eviction methods watch which tokens receive the least attention early on and drop them from the cache for good. Semantic compression methods group tokens into chunks during the initial read and replace each chunk with a single averaged summary before generation even starts. Both approaches work, and both share the same limitation, according to a team from the University of British Columbia and Microsoft Research studying long-context KV caching. A decision about what to keep gets made once, at the beginning, before the model has any idea which later question will actually depend on which earlier passage ",[355,371,372],{},[358,373,375],{"href":374},"#source-2","[2]",". Evidence that looked irrelevant in the first paragraph of a legal filing or a codebase might turn out to be the exact clause a later question hinges on, and once it has been evicted or blended into an average, there is no way back.",[348,378,379,380,362],{},"The researchers propose something they call SeKV, short for semantic KV cache, which keeps every span of text in two forms at once rather than picking one. A lightweight summary vector for each span stays resident on the GPU, cheap to scan and used only to decide whether that span looks relevant to the current decoding step. A separate, more detailed representation of the same span, built from a technique called singular value decomposition that factors a block of numbers into a smaller set of components capturing most of its structure, sits on the CPU instead, waiting to be fetched only if the summary suggests it is worth the trip. Segment boundaries themselves come from a signal already available for free during the model's first pass over the text, a measure of how surprising each token is given what came before it, which tends to spike at genuine topic shifts and drop within a coherent run of text ",[355,381,382],{},[358,383,375],{"href":374},[319,385,340,387,340,391,340,397],{"style":386},"width: 100%; margin: 20px 0;",[342,388],{"src":389,"alt":390,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1722080767146-0ae3a8ff46a0?w=1200&auto=format&fit=crop","A satellite view of a dark lake surrounded by mountains, showing broad shapes clearly but withholding finer ground-level detail until the view is zoomed in, analogous to a coarse memory summary that a model can request higher resolution from on demand",[392,393,396],"h3",{"style":394,"id":395},"margin: 1rem 0 0.5rem 0;","a-coarse-view-that-can-be-zoomed","A Coarse View That Can Be Zoomed",[348,398,400],{"style":399},"margin: 0;","A satellite photograph like this one shows enough to locate a shoreline or a mountain range, but not enough to make out a building or a road. Getting that level of detail means requesting a closer pass over one particular area, not over the whole frame. SeKV applies the same logic to context, routing over cheap summaries first and only reconstructing full detail for the handful of spans a given decoding step actually needs.",[348,402,403,404,408],{},"Whether a given span deserves that closer look gets decided by a small trained component, referred to as a zoom-in mechanism, that scores each summary against the model's current query and expands only the spans that clear a learned threshold. Across four long-context benchmarks, the reported result was a 5.9 percent average improvement over the strongest semantic-compression baseline tested, alongside a 53.3 percent reduction in GPU memory compared with keeping the full cache at a 128,000-token context length, achieved while adding fewer than 0.05 percent additional trainable parameters to a frozen base model ",[355,405,406],{},[358,407,375],{"href":374},". The efficiency gain and the accuracy gain showing up together is worth noting mainly because compression methods usually have to trade one for the other.",[331,410,412],{"id":411},"a-token-that-asks-for-a-refresh","A Token That Asks for a Refresh",[348,414,415,416,422],{},"A separate team, working across several Chinese research labs, tackled a closely related question from another angle. Rather than deciding what to keep once and reconstructing detail on demand, their system, called PReM for preserve and refresh memory, lets the model itself decide when its current compressed view of the context has gone stale and needs to be rebuilt ",[355,417,418],{},[358,419,421],{"href":420},"#source-3","[3]",". PReM trains a special memory token that the model can emit mid-generation. Emitting it triggers a fresh look back over the full context, re-selecting which chunks deserve to be kept at full token-level detail and which get replaced with an averaged stand-in, based on whatever the model is trying to figure out at that specific step rather than on whatever seemed important when the context was first read.",[348,424,425,426,430],{},"Training a model to condition its own generation on a memory state that changes partway through, rather than staying fixed for the whole response, required the researchers to split each generation step into two separate forward passes, one that handles the refresh decision and one that generates conditioned on the result. On a set of question-answering benchmarks using 32,000-token contexts, PReM reportedly outperformed eight established KV-cache and context-compression baselines under both 16 times and 32 times compression, improving average exact-match and F1 scores by up to 10.23 and 12.55 points respectively over the strongest baseline tested ",[355,427,428],{},[358,429,421],{"href":420},". A smaller 3-billion-parameter version of the model was also reported to outperform some larger 7-billion-parameter compression baselines, which the authors take as a sign that learning when to refresh can partly substitute for raw model scale, at least on the benchmarks tried so far.",[319,432,340,433,340,437,340,441],{"style":386},[342,434],{"src":435,"alt":436,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1573166364266-356ef04ae798?w=1200&auto=format&fit=crop","A person writing on a dry-erase whiteboard, part of the surface freshly wiped clean while other notes remain, analogous to a memory token that triggers a partial refresh of what a model keeps in view without discarding the underlying source material",[392,438,440],{"style":394,"id":439},"refreshing-the-board-without-losing-the-room","Refreshing the Board Without Losing the Room",[348,442,443],{"style":399},"Wiping part of a whiteboard clean does not erase the meeting that produced the notes, only the summary written down at the time. PReM's memory token works on a similar principle, refreshing the compressed view the model is currently attending to while the full source context stays available underneath, ready to be resummarized differently the next time a refresh is triggered.",[331,445,447],{"id":446},"memory-that-outlives-a-single-document","Memory That Outlives a Single Document",[348,449,450,451,457,458,462],{},"Both SeKV and PReM operate within a single long document or a single extended answer. A different line of work asks what changes when the relevant memory spans many separate conversations with the same user over time, a setting closer to how a deployed assistant actually gets used. A team centered at the University of Electronic Science and Technology of China proposes LightMem, which splits an agent's memory into three tiers, a short-term store for the immediate conversation, a mid-term store of reusable summaries from recent sessions, and a long-term store of consolidated knowledge, each managed by a small, purpose-built language model rather than a large one ",[355,452,453],{},[358,454,456],{"href":455},"#source-4","[4]",". Retrieval happens in two cheap stages, a coarse vector search followed by a semantic consistency check, before anything gets handed to the main model. Reported results on a long-term dialogue benchmark showed roughly a 2.5-point average gain in F1 score across model scales, with a median end-to-end latency around 581 milliseconds and retrieval alone completing in about 83 milliseconds ",[355,459,460],{},[358,461,456],{"href":455},". The appeal of the approach, from a systems perspective, sits less in the accuracy number and more in the latency figure, since memory operations that run on small models rather than the main large model avoid competing for the same expensive compute.",[348,464,465,466,472,473,362],{},"A related but distinct idea comes from a team at the University of Science and Technology of China, who point out that agents which compress a long document into memory through a single linear pass, reading start to finish and updating a running summary as they go, tend to prune evidence early that only turns out to matter once a much later part of the document has been read ",[355,467,468],{},[358,469,471],{"href":470},"#source-5","[5]",". Their system, ReMemR1, lets an agent issue what the authors call a callback query mid-reasoning, retrieving an earlier memory state instead of only ever moving forward through it, and trains the behavior with a reward signal that combines the final answer's correctness with denser, step-level feedback about whether a given callback was actually useful. On long-context question-answering benchmarks, the reported result was upward of a 20 percent relative reduction in error rate compared with linear memory baselines, with the retrieval mechanism adding less than 0.2 percent to overall computation time ",[355,474,475],{},[358,476,471],{"href":470},[331,478,480],{"id":479},"what-this-suggests","What This Suggests",[348,482,483],{},"Four independent groups converging on some version of the same idea, that compression decisions should stay reversible and resolution should be allocated on demand rather than fixed up front, is a pattern worth watching rather than a settled conclusion. Every result above comes from a specific benchmark and a specific context length, 32,000 tokens for PReM, 128,000 for SeKV, documents padded to a few thousand entries for ReMemR1, and none of them yet speaks directly to million-token agent contexts or to workloads where the CPU-to-GPU transfer that SeKV and similar systems depend on becomes a bottleneck of its own under real production load. Whether the accuracy gains reported here persist once these methods leave curated benchmarks and meet the messier, longer-horizon contexts that production agents actually accumulate remains an open question, and one that will likely need answering system by system rather than in the abstract.",[319,485,323,489,323,492],{"className":486},[487,488],"references","mt-8",[331,490,491],{"id":487},"References",[493,494,340,500,340,518,340,530,340,540,340,550,323],"ol",{"className":495},[496,497,498,499],"list-decimal","list-inside","space-y-2","mt-4",[501,502,504,505,509,510],"li",{"id":503},"source-1","R. Pope et al., \"Efficiently Scaling Transformer Inference,\" ",[506,507,508],"em",{},"Proceedings of the Sixth Conference on Machine Learning and Systems (MLSys 2023)",", 2023. DOI: ",[358,511,517],{"href":512,"target":513,"className":514},"https:\u002F\u002Fdoi.org\u002F10.48550\u002FarXiv.2211.05102","_blank",[515,516],"text-blue-600","underline","[Online]",[501,519,521,522,525,526],{"id":520},"source-2","A. Abaskohi et al., \"SeKV: Resolution-Adaptive KV Cache with Hierarchical Semantic Memory for Long-Context LLM Inference,\" ",[506,523,524],{},"arXiv",", 2026, ",[358,527,517],{"href":528,"target":513,"className":529},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.31145",[515,516],[501,531,533,534,525,536],{"id":532},"source-3","B. Yu et al., \"PReM: Learning What to Preserve and When to Refresh for Context Compression,\" ",[506,535,524],{},[358,537,517],{"href":538,"target":513,"className":539},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2607.14327",[515,516],[501,541,543,544,525,546],{"id":542},"source-4","J. Zhang et al., \"Lightweight LLM Agent Memory with Small Language Models,\" ",[506,545,524],{},[358,547,517],{"href":548,"target":513,"className":549},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.07798",[515,516],[501,551,553,554,556,557],{"id":552},"source-5","Y. Shi et al., \"Look Back to Reason Forward: Revisitable Memory for Long-Context LLM Agents,\" ",[506,555,524],{},", 2025, ",[358,558,517],{"href":559,"target":513,"className":560},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.23040",[515,516],{"title":562,"searchDepth":563,"depth":563,"links":564},"",2,[565,566,570,573,574,575],{"id":335,"depth":563,"text":336},{"id":365,"depth":563,"text":366,"children":567},[568],{"id":395,"depth":569,"text":396},3,{"id":411,"depth":563,"text":412,"children":571},[572],{"id":439,"depth":569,"text":440},{"id":446,"depth":563,"text":447},{"id":479,"depth":563,"text":480},{"id":487,"depth":563,"text":491},"2026-07-27","Long-context language models have relied on discarding or freezing most of what they read to keep memory costs down. New research on resolution-adaptive caching and refreshable memory tokens suggests a different approach, keeping everything at low resolution and reconstructing detail only when a later step actually needs it.","md",{"src":344},{"authors":581,"badge":587,"source":589},[582],{"avatar":583,"name":585,"to":586},{"src":584},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":588},"Long-Context Inference",{"name":590,"url":591},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":74,"description":577},"JZHObIsLZXHYqMm_rMP1pKlqCXS6qzRMaqUmKazEsmQ",[596,801],{"id":597,"title":242,"body":598,"date":789,"description":790,"extension":578,"image":791,"meta":792,"navigation":592,"path":243,"seo":799,"stem":244,"__hash__":800,"_path":243},"insights\u002Fnews\u002Finsights\u002Freasoning-budget-allocation.md",{"type":316,"value":599,"toc":776},[600,612,618,625,628,632,644,652,665,669,681,685,697,710,714,727,729,732],[319,601,323,603,323,607],{"className":602},[322],[325,604,242],{"className":605,"id":606},[328],"second-guessing-has-a-price",[331,608,611],{"className":609,"id":610},[334],"reasoning-models-are-learning-when-extra-compute-stops-helping","Reasoning Models Are Learning When Extra Compute Stops Helping",[319,613,340,614],{"style":339},[342,615],{"src":616,"alt":617,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1774660980287-420ea5b70d4f?w=1200&h=750&fit=crop&crop=focalpoint&fp-x=0.5&fp-y=0.62&auto=format","Hands reaching for a stack of poker chips on a green felt table, analogous to a reasoning model deciding whether another round of thinking is worth the compute it costs or whether the hand should be called as is",[348,619,620,621,362],{},"A poker player holding a decent hand faces a genuine decision each round. Calling costs chips now for a chance at a bigger pot later, and folding gives up whatever is already in play to avoid a loss that has not happened yet. Neither choice is obviously correct, and the right one depends on what is already known about the hand, not on how long the player stares at the cards. A team at the University of Science and Technology of China built something close to that same instinct into large reasoning models, the class of language models trained to generate long chains of intermediate reasoning steps before producing a final answer. Extending that reasoning costs compute, and the researchers argue the decision to keep going should depend on whether more thinking is likely to pay off, not on a fixed token budget handed down in advance ",[355,622,623],{},[358,624,375],{"href":374},[348,626,627],{},"Test-time compute, the amount of reasoning a model generates while answering a single query, has become one of the main levers for improving accuracy on hard problems, alongside training on more data or building bigger models. The general finding, replicated across many benchmarks, is that letting a model reason at length before answering tends to raise accuracy on tasks like competition mathematics. What has drawn less scrutiny is whether that relationship holds all the way up, or whether it eventually bends.",[331,629,631],{"id":630},"when-the-curve-bends-the-wrong-way","When the Curve Bends the Wrong Way",[348,633,634,635,639,640,362],{},"A team at Nanjing University set out to measure exactly that. Working with reasoning models forced to keep generating for a controlled number of tokens ranging from 500 up to 16,000, using a technique that appends the word Wait if the model tries to stop early, the researchers tracked how each additional block of reasoning tokens changed accuracy ",[355,636,637],{},[358,638,361],{"href":360},". Borrowing a concept from economics, the law of diminishing marginal returns, where each additional unit of input produces a smaller gain in output than the one before it, they found that the marginal benefit of extra reasoning tokens shrinks steadily as the budget grows, and for easier problems it can turn negative well before the budget runs out. Generating 8,000 tokens costs roughly sixteen times what 500 tokens costs, and much of that spending on simple problems buys nothing ",[355,641,642],{},[358,643,361],{"href":360},[348,645,646,647,651],{},"More strikingly, the researchers tracked individual problems through their reasoning trajectories and identified what they call flip events, moments where a model arrives at the correct answer relatively early, then keeps reasoning anyway and talks itself into a wrong one. Extended thinking was not simply wasted in these cases, it was actively harmful, with the model second-guessing an initial answer that had already been right ",[355,648,649],{},[358,650,361],{"href":360},". The severity of this pattern varied by model and by problem difficulty, easier problems flipped from correct to incorrect earlier in the reasoning trace than harder ones, which suggests that a single fixed thinking budget applied uniformly across every query is close to the worst way to spend a compute budget.",[319,653,340,654,340,658,340,662],{"style":386},[342,655],{"src":656,"alt":657,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1616556425595-93db848a50c7?w=1200&auto=format&fit=crop","Water pouring into a clear drinking glass that is already full, spilling over the rim, analogous to reasoning tokens added past the point where a model already has the right answer, where the extra pour does not help and can knock the answer over",[392,659,661],{"style":394,"id":660},"pouring-past-full","Pouring Past Full",[348,663,664],{"style":399},"Water poured into a glass that is already full does not raise the water level. It spills over the rim, and if the glass gets jostled in the process, some of what was already sitting there can end up on the table too. A model that keeps generating reasoning tokens well after it has settled on an answer is pouring past that same rim, spending compute that buys nothing, and every so often jostling loose a correct answer that would have stood if the pour had simply stopped.",[331,666,668],{"id":667},"betting-only-when-the-odds-are-good","Betting Only When the Odds Are Good",[348,670,671,672,676,677,362],{},"The USTC team's system, called Bet for budget-efficient thinking, tries to learn that same read on the table, when continued reasoning is worth its cost and when it is not. Rather than sizing the compute budget to how hard a problem looks on its surface, which the researchers point out is a poor proxy, Bet estimates how likely the model's current policy is to actually solve the problem if given more time, a quantity closer to solvability than to difficulty ",[355,673,674],{},[358,675,375],{"href":374},". A problem can look intimidating and still be within easy reach for a well-trained model, and a problem that reads simply can sit just outside what the model can currently do no matter how long it reasons. Based on this estimate, the system learns three behaviors described in the paper's own poker terms, a short solve that answers easy queries concisely, a nice fold that abstains early once continued reasoning shows near-zero expected return, and a hero call that commits substantial compute to problems that are hard but genuinely within reach. Trained with a two-stage process combining supervised examples with reinforcement learning under a cost-aware reward, Bet reportedly cut reasoning tokens by around 55 percent on average across seven benchmarks while improving overall accuracy, and the learned behavior transferred to scientific and logical reasoning tasks outside the mathematics domain it was trained on ",[355,678,679],{},[358,680,375],{"href":374},[331,682,684],{"id":683},"not-every-turn-deserves-the-same-budget","Not Every Turn Deserves the Same Budget",[348,686,687,688,692,693,362],{},"Most budget-allocation research, including the two studies above, treats a query as a single, self-contained reasoning problem. A team at Carnegie Mellon University points out that this framing breaks down once a model is having a multi-turn conversation, since a fixed per-turn budget spent generously on an easy opening question leaves less room for a much harder follow-up question three turns later ",[355,689,690],{},[358,691,421],{"href":420},". Their system, called TAB for turn-adaptive budgets, treats the whole conversation as a sequential compute allocation problem, formally a multi-objective decision process where the model has to decide, turn by turn, how much of a shared token budget to spend now versus hold in reserve for whatever comes later in the exchange. Trained with a group-based reinforcement learning method, TAB was reported to save up to 35 percent of tokens compared with static, turn-unaware budget policies while maintaining comparable accuracy on mathematical reasoning benchmarks, and a variant given advance knowledge of all the sub-questions in a conversation saved up to 40 percent ",[355,694,695],{},[358,696,421],{"href":420},[319,698,340,699,340,703,340,707],{"style":386},[342,700],{"src":701,"alt":702,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1769708115034-9bbe58bad782?w=1200&auto=format&fit=crop","Athletes clearing hurdles of a track and field race, some strides open and easy between barriers and others requiring a full jump, analogous to a multi-turn conversation where some turns are an easy stretch of track and others demand the model's full effort",[392,704,706],{"style":394,"id":705},"pacing-across-an-uneven-track","Pacing Across an Uneven Track",[348,708,709],{"style":399},"A hurdles race is not run at one constant effort. The open stretches between barriers call for a different pace than the barrier itself, and a runner who sprints flat out over every meter arrives at the tenth hurdle with nothing left to clear it. TAB spends its tokens the way that runner spends stride length, freely across the open stretches of a conversation, held back for whichever turn turns out to be the barrier.",[331,711,713],{"id":712},"teaching-the-habit-during-training-not-just-at-inference","Teaching the Habit During Training, Not Just at Inference",[348,715,716,717,721,722,726],{},"A fourth angle on the same problem moves the intervention earlier, into training itself. A framework called BACR, for budget-adaptive curriculum reasoning, argues that training a model under a single fixed token budget, or under budgets sampled uniformly at random, teaches it a compute habit that does not match the actual spread of problem difficulty it will face later ",[355,718,719],{},[358,720,456],{"href":455},". BACR conditions the model on its assigned budget as an explicit input during training, then uses a scheduler that shifts the distribution of training budgets from easy to hard as the model's own performance improves, paired with a reward that gives partial credit for intermediate reasoning steps even when a response gets cut off by the budget. On a set of mathematics benchmarks including MATH and AIME, the reported result was up to an 8.3 percent accuracy improvement under tight budgets alongside a 34 percent reduction in average token consumption compared with training under unconstrained reasoning ",[355,723,724],{},[358,725,456],{"href":455},". The gain under tight budgets specifically is the detail worth sitting with, since a curriculum that only helps when compute is already generous would be a much smaller contribution.",[331,728,480],{"id":479},[348,730,731],{},"None of these four systems agree on exactly where the compute-allocation decision should live. Bet puts it inside a single query, TAB puts it across the turns of a conversation, and BACR puts it back in training before any of that decision-making happens at inference time. Taken together they point toward the same underlying claim, that a uniform thinking budget is a specific and avoidable form of waste rather than a neutral default, and that the waste sometimes curdles into an actual accuracy cost through the flip events the Nanjing University team documented. Every reported gain here comes from mathematical reasoning benchmarks or benchmarks adjacent to them, and it remains an open question whether solvability estimates, turn-level budgeting, and curriculum scheduling generalize as cleanly to open-ended tasks like coding or long document analysis, where correctness is harder to check and a flip event is harder to even detect. From a systems perspective, the more durable lesson may be procedural rather than architectural, that any team deploying test-time compute at scale should be measuring the marginal accuracy per token spent, not just the accuracy at whatever budget happens to be the current default.",[319,733,323,735,323,737],{"className":734},[487,488],[331,736,491],{"id":487},[493,738,340,740,340,749,340,758,340,767,323],{"className":739},[496,497,498,499],[501,741,742,743,525,745],{"id":503},"S. Zhou et al., \"When More Thinking Hurts: Overthinking in LLM Test-Time Compute Scaling,\" ",[506,744,524],{},[358,746,517],{"href":747,"target":513,"className":748},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.10739",[515,516],[501,750,751,752,525,754],{"id":520},"Z. Zhou et al., \"Nice Fold or Hero Call: Learning Budget-Efficient Thinking for Adaptive Reasoning,\" ",[506,753,524],{},[358,755,517],{"href":756,"target":513,"className":757},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.11625",[515,516],[501,759,760,761,525,763],{"id":532},"N. Jali et al., \"Not All Turns Are Equally Hard: Adaptive Thinking Budgets for Efficient Multi-Turn Reasoning,\" ",[506,762,524],{},[358,764,517],{"href":765,"target":513,"className":766},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.05164",[515,516],[501,768,769,770,525,772],{"id":542},"A. Rahman et al., \"Avoiding Overthinking and Underthinking: Curriculum-Aware Budget Scheduling for LLMs,\" ",[506,771,524],{},[358,773,517],{"href":774,"target":513,"className":775},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.19780",[515,516],{"title":562,"searchDepth":563,"depth":563,"links":777},[778,779,782,783,786,787,788],{"id":610,"depth":563,"text":611},{"id":630,"depth":563,"text":631,"children":780},[781],{"id":660,"depth":569,"text":661},{"id":667,"depth":563,"text":668},{"id":683,"depth":563,"text":684,"children":784},[785],{"id":705,"depth":569,"text":706},{"id":712,"depth":563,"text":713},{"id":479,"depth":563,"text":480},{"id":487,"depth":563,"text":491},"2026-08-02","Reasoning models that think longer usually score higher, until they don't. New research on test-time compute shows extended reasoning can flip correct answers into wrong ones, and a cluster of recent papers proposes budgeting compute by solvability, by turn, and by training curriculum instead of by a fixed token limit.",{"src":616},{"authors":793,"badge":796,"source":798},[794],{"avatar":795,"name":585,"to":586},{"src":584},{"label":797},"Test-Time Compute",{"name":590,"url":591},{"title":242,"description":790},"aQTHbDPuptD50mePoROY3wl3C4CbsGKpNPxyRLhsVpY",{"id":802,"title":278,"body":803,"date":1025,"description":1026,"extension":578,"image":1027,"meta":1028,"navigation":592,"path":279,"seo":1035,"stem":280,"__hash__":1036,"_path":279},"insights\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival.md",{"type":316,"value":804,"toc":1012},[805,818,824,832,840,844,852,860,872,885,889,907,920,924,937,941,949,951,954],[319,806,323,808,323,813],{"className":807},[322],[325,809,812],{"className":810,"id":811},[328],"when-replay-is-not-an-option","When Replay Is Not an Option",[331,814,817],{"className":815,"id":816},[334],"streaming-q-learning-and-sarsa-get-a-second-look","Streaming Q-Learning and SARSA Get a Second Look",[319,819,340,820],{"style":339},[342,821],{"src":822,"alt":823,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1772843245076-cb1b1136ceb8?w=1200&auto=format&fit=crop","A fast-flowing forest river passing over rocks without pooling, analogous to streaming reinforcement learning where each experience is used once as it arrives and never stored for later replay",[348,825,826,827,831],{},"A river never pools before it moves on. Water arrives from upstream, passes a given point exactly once, and continues toward the sea without waiting to be collected first. Reinforcement learning, the branch of machine learning where a software agent learns by acting in an environment and receiving rewards, was originally built the same way. Q-learning and SARSA, two classic algorithms that both estimate the future payoff of taking a given action in a given situation, were designed to update from one experience at a time and move on, a mode of operation known as streaming or online learning ",[355,828,829],{},[358,830,375],{"href":374},". Both trace back to value iteration, the older dynamic-programming idea of repeatedly correcting a value estimate until it stops changing.",[348,833,834,835,839],{},"Modern deep reinforcement learning rarely runs this way anymore. Since Deep Q-Networks combined Q-learning with neural networks to reach competitive play on Atari games, most systems have leaned on a replay buffer, a large memory bank that stores past experience so it can be resampled again and again in batches ",[355,836,837],{},[358,838,361],{"href":360},". Averaging many samples together before each update smooths out noise and lets the same experience teach the network more than once. It works well, but it assumes there is somewhere to put the water while it waits its turn.",[331,841,843],{"id":842},"why-the-river-stopped-flowing","Why the River Stopped Flowing",[348,845,846,847,851],{},"Not every system can afford that assumption. A small robot with a modest onboard processor, a wearable sensor, or a system bound by strict data-privacy limits may not have the memory, bandwidth, or permission to keep raw experience around for later replay ",[355,848,849],{},[358,850,375],{"href":374},". When the water cannot be stored, learning has to happen as it passes through, or not happen at all.",[348,853,854,855,859],{},"When researchers tried simply removing the replay buffer from standard deep reinforcement learning algorithms, the results were not encouraging. Learning became unstable or collapsed outright across several benchmark tasks, an effect one recent paper terms the stream barrier ",[355,856,857],{},[358,858,375],{"href":374},". Something about updating incrementally, without the averaging effect of a batch, exposed weaknesses that batch learning had quietly been covering up. Both Q-learning and SARSA compute what is called a temporal difference error, the gap between what the network predicted and what actually happened one step later, then nudge the estimate a small amount toward closing that gap. In a deep network, that nudge is scaled by a step size, and getting the step size wrong turns out to matter far more without a batch to soften the blow.",[348,861,862,863,867,868,362],{},"One proposed fix combines several older ideas that had fallen out of fashion. Eligibility traces, a short-term memory that decays gradually and lets a single reward influence not just the most recent action but several that came before it, were reintroduced alongside layer normalization, a technique that keeps a neural network's internal activity at a stable scale over time, and a sparse pattern of initial connections that reduces interference between unrelated inputs ",[355,864,865],{},[358,866,375],{"href":374},". Bundled into algorithms called stream Q, stream SARSA, and stream actor-critic, these techniques reportedly let a network learn from Atari games, robotic control benchmarks, and a real-world electricity demand forecasting task as each experience arrives, matching or outperforming batch methods on several of them ",[355,869,870],{},[358,871,375],{"href":374},[319,873,340,874,340,878,340,882],{"style":386},[342,875],{"src":876,"alt":877,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1780034766211-bca35b171cd4?w=1200&auto=format&fit=crop","An industrial valve wheel mounted on a pipe, analogous to choosing a learning step size by feel, where turning too far risks a burst pipe and turning too little changes nothing",[392,879,881],{"style":394,"id":880},"tuning-the-flow-without-a-gauge","Tuning the Flow Without a Gauge",[348,883,884],{"style":399},"A valve wheel like this carries no readout of how much water is actually moving through the pipe behind it. Whoever turns it has to judge the right amount by feel, guided mostly by what happened the last time it was turned this far. Step size, the setting that controls how large a correction a learning algorithm makes after each mistake, has traditionally been chosen the same way in streaming reinforcement learning. Too far, and the update destabilizes the model. Too little, and nothing changes fast enough to matter.",[331,886,888],{"id":887},"choosing-the-step-size-on-purpose","Choosing the Step Size on Purpose",[348,890,891,892,896,897,901,902,906],{},"A separate line of work asks a different question about that same wheel. Rather than picking a step size measured in the units of a network's internal weights, why not decide first how much the model's actual prediction should change, then solve backward for whichever step size makes that happen, sidestepping the guesswork that made the stream barrier so damaging in the first place ",[355,893,894],{},[358,895,421],{"href":420},". Researchers working alongside Richard Sutton, one of the field's founding figures, call the result intentional updates. Instead of fixing how far to turn the wheel, the method fixes how much water should flow and calculates the turn needed to get there ",[355,898,899],{},[358,900,421],{"href":420},". Applied to temporal difference learning and to Q-learning, the approach is reported to reach streaming performance that is often comparable to batch and replay-buffer methods across the domains tested so far ",[355,903,904],{},[358,905,421],{"href":420},". Whether that holds up on tasks with much longer horizons than the current benchmarks is a question the authors leave open, and one further work will need to settle.",[319,908,340,909,340,913,340,917],{"style":386},[342,910],{"src":911,"alt":912,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1611663806011-b37e091090f0?w=1200&auto=format&fit=crop","A small green microcontroller development board with visible chips and pins, analogous to a resource-constrained device that has room to hold only the current moment of experience, not a warehouse of stored samples",[392,914,916],{"style":394,"id":915},"learning-on-the-device-itself","Learning on the Device Itself",[348,918,919],{"style":399},"A board this size holds only what is happening right now, with no spare memory set aside for a warehouse of stored samples. That is not an edge case for streaming reinforcement learning so much as close to the point of the exercise.",[331,921,923],{"id":922},"from-simulation-to-real-hardware","From Simulation to Real Hardware",[348,925,926,927,931,932,936],{},"Researchers recently tested an incremental policy gradient method, a family of algorithms that adjust a decision-making policy directly rather than through an intermediate value estimate, under the name Action Value Gradient. On simulated robotic control benchmarks, it was reportedly the only incremental method among those compared that learned effectively without a replay buffer, target network, or batch update ",[355,928,929],{},[358,930,456],{"href":455},". It was later used to train a robotic manipulator arm and a mobile robot with only real-time, incremental updates, which the authors describe as the first demonstration of effective deep reinforcement learning on physical robots restricted to incremental learning ",[355,933,934],{},[358,935,456],{"href":455},". The result matters less because of the specific robots involved and more because it suggests the resource savings from dropping batch machinery do not necessarily come at the cost of learning ability, at least on the hardware and tasks tested.",[331,938,940],{"id":939},"where-the-simpler-algorithms-already-work","Where the Simpler Algorithms Already Work",[348,942,943,944,948],{},"None of this implies classical Q-learning and SARSA need deep networks or streaming tricks to be useful today. In a 2025 comparison of microgrid energy management, a setting involving batteries, solar generation, and shifting demand, researchers found that plain tabular Q-learning and SARSA still produced meaningful cost savings compared with having no learning agent at all, even though a deep Q-network outperformed both by a further margin ",[355,945,946],{},[358,947,471],{"href":470},". Whether that gap between simple and deep methods narrows or widens as more real energy systems get instrumented is a question best left to people closer to that domain. From a systems perspective, though, it is a useful reminder that the current streaming revival is not solving a problem that never existed before. It is trying to bring the modeling capacity of deep networks into settings where the older, simpler algorithms already had to operate without a replay buffer by design.",[331,950,480],{"id":479},[348,952,953],{},"Streaming reinforcement learning was not so much invented recently as recovered. Q-learning and SARSA were streaming algorithms from the outset, descendants of value iteration's habit of repeatedly correcting an estimate against a moving target. Batch learning became dominant afterward, once neural networks entered the picture, partly because batching hid instabilities that incremental updates expose immediately. The recent work on the stream barrier, intentional step sizes, and incremental policy gradients does not eliminate that instability. It builds tools aimed at the constraint that made replay buffers attractive in the first place, limited onboard hardware and experience that can only be used once. Whether these methods extend cleanly to longer, messier tasks than the current benchmarks cover remains an open question, one likely to be answered one small robot and one edge device at a time.",[319,955,323,957,323,959],{"className":956},[487,488],[331,958,491],{"id":487},[493,960,340,962,340,973,340,983,340,992,340,1001,323],{"className":961},[496,497,498,499],[501,963,964,965,968,969],{"id":503},"V. Mnih et al., \"Human-level control through deep reinforcement learning,\" ",[506,966,967],{},"Nature",", vol. 518, no. 7540, pp. 529–533, 2015. DOI: ",[358,970,517],{"href":971,"target":513,"className":972},"https:\u002F\u002Fdoi.org\u002F10.1038\u002Fnature14236",[515,516],[501,974,975,976,978,979],{"id":520},"M. Elsayed et al., \"Streaming Deep Reinforcement Learning Finally Works,\" ",[506,977,524],{},", 2024, ",[358,980,517],{"href":981,"target":513,"className":982},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2410.14606",[515,516],[501,984,985,986,525,988],{"id":532},"A. Sharifnassab et al., \"Intentional Updates for Streaming Reinforcement Learning,\" ",[506,987,524],{},[358,989,517],{"href":990,"target":513,"className":991},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.19033",[515,516],[501,993,994,995,978,997],{"id":542},"G. Vasan et al., \"Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers,\" ",[506,996,524],{},[358,998,517],{"href":999,"target":513,"className":1000},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2411.15370",[515,516],[501,1002,1003,1004,1007,1008],{"id":552},"S. Ramesh et al., \"Comparative analysis of Q-learning, SARSA, and deep Q-network for microgrid energy management,\" ",[506,1005,1006],{},"Scientific Reports",", vol. 15, article 694, 2025. DOI: ",[358,1009,517],{"href":1010,"target":513,"className":1011},"https:\u002F\u002Fdoi.org\u002F10.1038\u002Fs41598-024-83625-8",[515,516],{"title":562,"searchDepth":563,"depth":563,"links":1013},[1014,1015,1018,1021,1022,1023,1024],{"id":816,"depth":563,"text":817},{"id":842,"depth":563,"text":843,"children":1016},[1017],{"id":880,"depth":569,"text":881},{"id":887,"depth":563,"text":888,"children":1019},[1020],{"id":915,"depth":569,"text":916},{"id":922,"depth":563,"text":923},{"id":939,"depth":563,"text":940},{"id":479,"depth":563,"text":480},{"id":487,"depth":563,"text":491},"2026-07-19","New research revisits streaming Q-learning and SARSA, the original one-sample-at-a-time reinforcement learning algorithms, examining why deep versions became unstable without a replay buffer and what recent step-size and eligibility trace fixes suggest for on-device learning.",{"src":822},{"authors":1029,"badge":1032,"source":1034},[1030],{"avatar":1031,"name":585,"to":586},{"src":584},{"label":1033},"Reinforcement Learning",{"name":590,"url":591},{"title":278,"description":1026},"x71f018zPUk62pRKT3W7CaR_Z0cFN4ccOwkGaz2NRDg",1787055018842]