[{"data":1,"prerenderedAt":1049},["ShallowReactive",2],{"navigation":3,"\u002Fnews\u002Finsights\u002Freasoning-budget-allocation":313,"\u002Fnews\u002Finsights\u002Freasoning-budget-allocation-surround":578},[4,8,17,21,25,29,33,37,301,305,309],{"title":5,"path":6,"stem":7},"About Thinkata Intelligence","\u002Fabout","about",{"title":9,"path":10,"stem":11,"children":12},"Authentication","\u002Fauth","auth",[13],{"title":14,"path":15,"stem":16},"Email Confirmation","\u002Fauth\u002Fconfirmation","auth\u002Fconfirmation",{"title":18,"path":19,"stem":20},"Case Studies","\u002Fcase-studies","case-studies",{"title":22,"path":23,"stem":24},"Contact Us","\u002Fcontact","contact",{"title":26,"path":27,"stem":28},"Thinkata - Advanced AI Engineering & Multi-Agent System Solutions","\u002F","index",{"title":30,"path":31,"stem":32},"Insights","\u002Finsights","insights",{"title":34,"path":35,"stem":36},"Leadership","\u002Fleadership","leadership",{"title":38,"path":39,"stem":40,"children":41},"News","\u002Fnews","news",[42,45,69],{"title":43,"path":39,"stem":44},"News & Insights","news\u002Findex",{"title":18,"path":46,"stem":47,"children":48},"\u002Fnews\u002Fcase-studies","news\u002Fcase-studies",[49,53,57,61,65],{"title":50,"path":51,"stem":52},"Building Secure and Scalable AI Infrastructure: Integrating with Existing Systems through Modern Cloud Frameworks","\u002Fnews\u002Fcase-studies\u002Fcloud-infrastructure-ai","news\u002Fcase-studies\u002Fcloud-infrastructure-ai",{"title":54,"path":55,"stem":56},"Making Sense of Financial Regulations: How AI Teams Can Tackle Complex Documents","\u002Fnews\u002Fcase-studies\u002Ffinancial-regulations","news\u002Fcase-studies\u002Ffinancial-regulations",{"title":58,"path":59,"stem":60},"AI-Powered Transformations in Healthcare","\u002Fnews\u002Fcase-studies\u002Fhealth-care","news\u002Fcase-studies\u002Fhealth-care",{"title":62,"path":63,"stem":64},"Generative AI in Upstream Natural Gas: Shell's Exploration Initiative","\u002Fnews\u002Fcase-studies\u002Foil-gas","news\u002Fcase-studies\u002Foil-gas",{"title":66,"path":67,"stem":68},"Optimizing Manufacturing with AI-Driven Multi-Agent Systems","\u002Fnews\u002Fcase-studies\u002Fsupply-chain-optimization","news\u002Fcase-studies\u002Fsupply-chain-optimization",{"title":30,"path":70,"stem":71,"children":72},"\u002Fnews\u002Finsights","news\u002Finsights",[73,77,81,85,89,93,97,101,105,109,113,117,121,125,129,133,137,141,145,149,153,157,161,165,169,173,177,181,185,189,193,197,201,205,209,213,217,221,225,229,233,237,241,245,249,253,257,261,265,269,273,277,281,285,289,293,297],{"title":74,"path":75,"stem":76},"Nothing Gets Deleted, Just Blurred","\u002Fnews\u002Finsights\u002Fadaptive-context-memory","news\u002Finsights\u002Fadaptive-context-memory",{"title":78,"path":79,"stem":80},"The Capability-Reliability Split in Agent Systems","\u002Fnews\u002Finsights\u002Fagent-capability-reliability-split","news\u002Finsights\u002Fagent-capability-reliability-split",{"title":82,"path":83,"stem":84},"The Rise of AI Agents in Cyberattacks: Latest Research and Threats","\u002Fnews\u002Finsights\u002Fai-agent-cyber-threats","news\u002Finsights\u002Fai-agent-cyber-threats",{"title":86,"path":87,"stem":88},"Winning the Bid Is Not the Same as Knowing the Price","\u002Fnews\u002Finsights\u002Fai-agent-economics","news\u002Finsights\u002Fai-agent-economics",{"title":90,"path":91,"stem":92},"The Smart Enterprise AI Stack: Why Teams of AI Agents Beat Solo Models Consistently","\u002Fnews\u002Finsights\u002Fai-architecture","news\u002Finsights\u002Fai-architecture",{"title":94,"path":95,"stem":96},"When Seeing Everything Becomes the Only Option","\u002Fnews\u002Finsights\u002Fai-comprehensive-observability","news\u002Finsights\u002Fai-comprehensive-observability",{"title":98,"path":99,"stem":100},"The Data Infrastructure AI-Native Systems Can't Ignore","\u002Fnews\u002Finsights\u002Fai-data-layer","news\u002Finsights\u002Fai-data-layer",{"title":102,"path":103,"stem":104},"Enterprise AI Triage Systems: Intelligent Automation for Large-Scale Operations","\u002Fnews\u002Finsights\u002Fai-enterprise-triage","news\u002Finsights\u002Fai-enterprise-triage",{"title":106,"path":107,"stem":108},"When Oversight Becomes Infrastructure","\u002Fnews\u002Finsights\u002Fai-governed-autonomy","news\u002Finsights\u002Fai-governed-autonomy",{"title":110,"path":111,"stem":112},"Designing for Graceful Failure in Compound AI Systems","\u002Fnews\u002Finsights\u002Fai-graceful-failure","news\u002Finsights\u002Fai-graceful-failure",{"title":114,"path":115,"stem":116},"Intelligent Composability: Building AI Systems Like Orchestra, Not Soloists","\u002Fnews\u002Finsights\u002Fai-intelligent-composability","news\u002Finsights\u002Fai-intelligent-composability",{"title":118,"path":119,"stem":120},"Building the Plane While Flying It — Migrating from Monolith to AI-Native Without Stopping","\u002Fnews\u002Finsights\u002Fai-migration-path","news\u002Finsights\u002Fai-migration-path",{"title":122,"path":123,"stem":124},"Stability Through Continuous Adaptation","\u002Fnews\u002Finsights\u002Fai-native-overview","news\u002Finsights\u002Fai-native-overview",{"title":126,"path":127,"stem":128},"Provable Stability: Mathematical Guarantees for Adaptive AI Systems","\u002Fnews\u002Finsights\u002Fai-provable-stability","news\u002Finsights\u002Fai-provable-stability",{"title":130,"path":131,"stem":132},"How Temperature Tuning Makes or Breaks Reinforcement Learning","\u002Fnews\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse","news\u002Finsights\u002Fai-soft-actor-critic-entropy-collapse",{"title":134,"path":135,"stem":136},"Testing What Can't Be Predicted","\u002Fnews\u002Finsights\u002Fai-systems-testing","news\u002Finsights\u002Fai-systems-testing",{"title":138,"path":139,"stem":140},"Delegating the Job Is Not the Same as Delegating the Rules","\u002Fnews\u002Finsights\u002Fauthorization-drift-multi-agent","news\u002Finsights\u002Fauthorization-drift-multi-agent",{"title":142,"path":143,"stem":144},"Closing the Loop: How Human Corrections Can Make AI Systems Smarter Over Time","\u002Fnews\u002Finsights\u002Fclosing-the-loop","news\u002Finsights\u002Fclosing-the-loop",{"title":146,"path":147,"stem":148},"Multi-Path Reasoning: Collaborative and Competitive Approaches in AI","\u002Fnews\u002Finsights\u002Fcollaborative-competitive-agents","news\u002Finsights\u002Fcollaborative-competitive-agents",{"title":150,"path":151,"stem":152},"Why Challenges Supercharge Smarts for Humans and AI","\u002Fnews\u002Finsights\u002Fcompetition-improves-ai","news\u002Finsights\u002Fcompetition-improves-ai",{"title":154,"path":155,"stem":156},"Context is Infrastructure, Not Instructions","\u002Fnews\u002Finsights\u002Fcontext-is-infrastructure","news\u002Finsights\u002Fcontext-is-infrastructure",{"title":158,"path":159,"stem":160},"Context is the New Code","\u002Fnews\u002Finsights\u002Fcontext-is-new-code","news\u002Finsights\u002Fcontext-is-new-code",{"title":162,"path":163,"stem":164},"Continuous Thought Machines","\u002Fnews\u002Finsights\u002Fcontinuous-thought-machines","news\u002Finsights\u002Fcontinuous-thought-machines",{"title":166,"path":167,"stem":168},"Don't Vibe, Architect","\u002Fnews\u002Finsights\u002Fdont-vibe-architect","news\u002Finsights\u002Fdont-vibe-architect",{"title":170,"path":171,"stem":172},"The Edge of the Underdefined","\u002Fnews\u002Finsights\u002Fedge-of-the-underdefined","news\u002Finsights\u002Fedge-of-the-underdefined",{"title":174,"path":175,"stem":176},"Experts All the Way Down","\u002Fnews\u002Finsights\u002Fexperts-all-the-way","news\u002Finsights\u002Fexperts-all-the-way",{"title":178,"path":179,"stem":180},"A Multi-Tier Safety Architecture for Critical Applications","\u002Fnews\u002Finsights\u002Ffour-tier-architecture","news\u002Finsights\u002Ffour-tier-architecture",{"title":182,"path":183,"stem":184},"Green Dashboard, Unhappy Users","\u002Fnews\u002Finsights\u002Fgreen-dashboard-unhappy-users","news\u002Finsights\u002Fgreen-dashboard-unhappy-users",{"title":186,"path":187,"stem":188},"Hybrid Autoregressive Residual Tokens","\u002Fnews\u002Finsights\u002Fhart-model","news\u002Finsights\u002Fhart-model",{"title":190,"path":191,"stem":192},"Hierarchical Reasoning in Artificial Intelligence","\u002Fnews\u002Finsights\u002Fhierarchical-approaches","news\u002Finsights\u002Fhierarchical-approaches",{"title":194,"path":195,"stem":196},"Latent Diffusion for Language Generation: A Comprehensive Overview","\u002Fnews\u002Finsights\u002Flatent-diffusion-for-language","news\u002Finsights\u002Flatent-diffusion-for-language",{"title":198,"path":199,"stem":200},"Breaking Language Barriers: How AI Can Translate Without Examples","\u002Fnews\u002Finsights\u002Flearning-languages","news\u002Finsights\u002Flearning-languages",{"title":202,"path":203,"stem":204},"The Emergence of AI Deception: How Large Language Models Have Learned to Strategically Mislead Users","\u002Fnews\u002Finsights\u002Fllm-deception","news\u002Finsights\u002Fllm-deception",{"title":206,"path":207,"stem":208},"Grading on a Shared Curve","\u002Fnews\u002Finsights\u002Fllm-judge-correlated-errors","news\u002Finsights\u002Fllm-judge-correlated-errors",{"title":210,"path":211,"stem":212},"Synergizing Specialized Reasoning and General Capabilities in AI","\u002Fnews\u002Finsights\u002Fllm-reasoning-advances","news\u002Finsights\u002Fllm-reasoning-advances",{"title":214,"path":215,"stem":216},"The Expensive Default","\u002Fnews\u002Finsights\u002Fllm-routing-cost-quality","news\u002Finsights\u002Fllm-routing-cost-quality",{"title":218,"path":219,"stem":220},"The AI That Rewrites Itself: MIT's Breakthrough in Self-Adapting Language Models","\u002Fnews\u002Finsights\u002Fllm-seal","news\u002Finsights\u002Fllm-seal",{"title":222,"path":223,"stem":224},"Metacognitive Reinforcement Learning for Self-Improving AI Systems","\u002Fnews\u002Finsights\u002Fmetacognitive-reinforcement-learning","news\u002Finsights\u002Fmetacognitive-reinforcement-learning",{"title":226,"path":227,"stem":228},"Revolutionary Advancements in Mixture of Experts (MoE) Architectures","\u002Fnews\u002Finsights\u002Fmixture-of-experts","news\u002Finsights\u002Fmixture-of-experts",{"title":230,"path":231,"stem":232},"One Model, Many Customers, and the Leak Nobody Tests For","\u002Fnews\u002Finsights\u002Fmulti-tenant-ai-isolation","news\u002Finsights\u002Fmulti-tenant-ai-isolation",{"title":234,"path":235,"stem":236},"Balancing Neural Plasticity and Stability","\u002Fnews\u002Finsights\u002Fneural-plasticity","news\u002Finsights\u002Fneural-plasticity",{"title":238,"path":239,"stem":240},"Offline RL and the Data Flywheel","\u002Fnews\u002Finsights\u002Foffline-rl-data-flywheel","news\u002Finsights\u002Foffline-rl-data-flywheel",{"title":242,"path":243,"stem":244},"Second-Guessing Has a Price","\u002Fnews\u002Finsights\u002Freasoning-budget-allocation","news\u002Finsights\u002Freasoning-budget-allocation",{"title":246,"path":247,"stem":248},"Reasoning You Can Check","\u002Fnews\u002Finsights\u002Freasoning-you-can-check","news\u002Finsights\u002Freasoning-you-can-check",{"title":250,"path":251,"stem":252},"When Optimization Optimizes Itself","\u002Fnews\u002Finsights\u002Frecursive-goodhart","news\u002Finsights\u002Frecursive-goodhart",{"title":254,"path":255,"stem":256},"Reward Design as Architecture","\u002Fnews\u002Finsights\u002Freward-design-as-architecture","news\u002Finsights\u002Freward-design-as-architecture",{"title":258,"path":259,"stem":260},"When Success Has No Author: The Temporal Credit Assignment Problem","\u002Fnews\u002Finsights\u002Frl-credit-assignment-problem","news\u002Finsights\u002Frl-credit-assignment-problem",{"title":262,"path":263,"stem":264},"Beyond Entropy Collapse: When Exploration Succeeds but Learning Fails","\u002Fnews\u002Finsights\u002Frl-optimization-gaps","news\u002Finsights\u002Frl-optimization-gaps",{"title":266,"path":267,"stem":268},"The Path to Practical Confidential Computing for AI Systems","\u002Fnews\u002Finsights\u002Fsecure-ai-architectures","news\u002Finsights\u002Fsecure-ai-architectures",{"title":270,"path":271,"stem":272},"Guess First, Check Later","\u002Fnews\u002Finsights\u002Fspeculative-execution-pattern","news\u002Finsights\u002Fspeculative-execution-pattern",{"title":274,"path":275,"stem":276},"Spiking Neural Networks for Energy-Efficient AI","\u002Fnews\u002Finsights\u002Fspiking-neural-networks","news\u002Finsights\u002Fspiking-neural-networks",{"title":278,"path":279,"stem":280},"When Replay Is Not an Option: Streaming Q-Learning and SARSA Get a Second Look","\u002Fnews\u002Finsights\u002Fstreaming-q-learning-revival","news\u002Finsights\u002Fstreaming-q-learning-revival",{"title":282,"path":283,"stem":284},"The Turn as the Unit of Quality","\u002Fnews\u002Finsights\u002Fstructured-iteration-quality","news\u002Finsights\u002Fstructured-iteration-quality",{"title":286,"path":287,"stem":288},"AI Speech Translation: Breaking Down Language Barriers","\u002Fnews\u002Finsights\u002Fsts-performance-advances","news\u002Finsights\u002Fsts-performance-advances",{"title":290,"path":291,"stem":292},"Test-Time Training Layers: The Next Evolution in Transformer Architecture","\u002Fnews\u002Finsights\u002Ftest-time-training-layers","news\u002Finsights\u002Ftest-time-training-layers",{"title":294,"path":295,"stem":296},"Breakthrough: Large Language Models Pass the Turing Test","\u002Fnews\u002Finsights\u002Fturing-tests","news\u002Finsights\u002Fturing-tests",{"title":298,"path":299,"stem":300},"Training in a World That Does Not Exist Yet","\u002Fnews\u002Finsights\u002Fworld-models-as-infrastructure","news\u002Finsights\u002Fworld-models-as-infrastructure",{"title":302,"path":303,"stem":304},"Privacy Policy","\u002Fprivacy","privacy",{"title":306,"path":307,"stem":308},"Research","\u002Fresearch","research",{"title":310,"path":311,"stem":312},"Terms of Service","\u002Fterms","terms",{"id":314,"title":242,"body":315,"date":559,"description":560,"extension":561,"image":562,"meta":563,"navigation":575,"path":243,"seo":576,"stem":244,"__hash__":577},"insights\u002Fnews\u002Finsights\u002Freasoning-budget-allocation.md",{"type":316,"value":317,"toc":543},"minimark",[318,337,347,360,363,367,381,389,406,410,422,426,440,453,457,472,476,479],[319,320,323,324,323,330],"div",{"className":321},[322],"page-title","\n  ",[325,326,242],"h1",{"className":327,"id":329},[328],"page-title__main","second-guessing-has-a-price",[331,332,336],"h2",{"className":333,"id":335},[334],"page-title__sub","reasoning-models-are-learning-when-extra-compute-stops-helping","Reasoning Models Are Learning When Extra Compute Stops Helping",[319,338,340,341],{"style":339},"width: 100%; padding: 2%;","\n    ",[342,343],"img",{"src":344,"alt":345,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1774660980287-420ea5b70d4f?w=1200&h=750&fit=crop&crop=focalpoint&fp-x=0.5&fp-y=0.62&auto=format","Hands reaching for a stack of poker chips on a green felt table, analogous to a reasoning model deciding whether another round of thinking is worth the compute it costs or whether the hand should be called as is","width: 100%; height: auto;",[348,349,350,351,359],"p",{},"A poker player holding a decent hand faces a genuine decision each round. Calling costs chips now for a chance at a bigger pot later, and folding gives up whatever is already in play to avoid a loss that has not happened yet. Neither choice is obviously correct, and the right one depends on what is already known about the hand, not on how long the player stares at the cards. A team at the University of Science and Technology of China built something close to that same instinct into large reasoning models, the class of language models trained to generate long chains of intermediate reasoning steps before producing a final answer. Extending that reasoning costs compute, and the researchers argue the decision to keep going should depend on whether more thinking is likely to pay off, not on a fixed token budget handed down in advance ",[352,353,354],"sup",{},[355,356,358],"a",{"href":357},"#source-2","[2]",".",[348,361,362],{},"Test-time compute, the amount of reasoning a model generates while answering a single query, has become one of the main levers for improving accuracy on hard problems, alongside training on more data or building bigger models. The general finding, replicated across many benchmarks, is that letting a model reason at length before answering tends to raise accuracy on tasks like competition mathematics. What has drawn less scrutiny is whether that relationship holds all the way up, or whether it eventually bends.",[331,364,366],{"id":365},"when-the-curve-bends-the-wrong-way","When the Curve Bends the Wrong Way",[348,368,369,370,376,377,359],{},"A team at Nanjing University set out to measure exactly that. Working with reasoning models forced to keep generating for a controlled number of tokens ranging from 500 up to 16,000, using a technique that appends the word Wait if the model tries to stop early, the researchers tracked how each additional block of reasoning tokens changed accuracy ",[352,371,372],{},[355,373,375],{"href":374},"#source-1","[1]",". Borrowing a concept from economics, the law of diminishing marginal returns, where each additional unit of input produces a smaller gain in output than the one before it, they found that the marginal benefit of extra reasoning tokens shrinks steadily as the budget grows, and for easier problems it can turn negative well before the budget runs out. Generating 8,000 tokens costs roughly sixteen times what 500 tokens costs, and much of that spending on simple problems buys nothing ",[352,378,379],{},[355,380,375],{"href":374},[348,382,383,384,388],{},"More strikingly, the researchers tracked individual problems through their reasoning trajectories and identified what they call flip events, moments where a model arrives at the correct answer relatively early, then keeps reasoning anyway and talks itself into a wrong one. Extended thinking was not simply wasted in these cases, it was actively harmful, with the model second-guessing an initial answer that had already been right ",[352,385,386],{},[355,387,375],{"href":374},". The severity of this pattern varied by model and by problem difficulty, easier problems flipped from correct to incorrect earlier in the reasoning trace than harder ones, which suggests that a single fixed thinking budget applied uniformly across every query is close to the worst way to spend a compute budget.",[319,390,340,392,340,396,340,402],{"style":391},"width: 100%; margin: 20px 0;",[342,393],{"src":394,"alt":395,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1616556425595-93db848a50c7?w=1200&auto=format&fit=crop","Water pouring into a clear drinking glass that is already full, spilling over the rim, analogous to reasoning tokens added past the point where a model already has the right answer, where the extra pour does not help and can knock the answer over",[397,398,401],"h3",{"style":399,"id":400},"margin: 1rem 0 0.5rem 0;","pouring-past-full","Pouring Past Full",[348,403,405],{"style":404},"margin: 0;","Water poured into a glass that is already full does not raise the water level. It spills over the rim, and if the glass gets jostled in the process, some of what was already sitting there can end up on the table too. A model that keeps generating reasoning tokens well after it has settled on an answer is pouring past that same rim, spending compute that buys nothing, and every so often jostling loose a correct answer that would have stood if the pour had simply stopped.",[331,407,409],{"id":408},"betting-only-when-the-odds-are-good","Betting Only When the Odds Are Good",[348,411,412,413,417,418,359],{},"The USTC team's system, called Bet for budget-efficient thinking, tries to learn that same read on the table, when continued reasoning is worth its cost and when it is not. Rather than sizing the compute budget to how hard a problem looks on its surface, which the researchers point out is a poor proxy, Bet estimates how likely the model's current policy is to actually solve the problem if given more time, a quantity closer to solvability than to difficulty ",[352,414,415],{},[355,416,358],{"href":357},". A problem can look intimidating and still be within easy reach for a well-trained model, and a problem that reads simply can sit just outside what the model can currently do no matter how long it reasons. Based on this estimate, the system learns three behaviors described in the paper's own poker terms, a short solve that answers easy queries concisely, a nice fold that abstains early once continued reasoning shows near-zero expected return, and a hero call that commits substantial compute to problems that are hard but genuinely within reach. Trained with a two-stage process combining supervised examples with reinforcement learning under a cost-aware reward, Bet reportedly cut reasoning tokens by around 55 percent on average across seven benchmarks while improving overall accuracy, and the learned behavior transferred to scientific and logical reasoning tasks outside the mathematics domain it was trained on ",[352,419,420],{},[355,421,358],{"href":357},[331,423,425],{"id":424},"not-every-turn-deserves-the-same-budget","Not Every Turn Deserves the Same Budget",[348,427,428,429,435,436,359],{},"Most budget-allocation research, including the two studies above, treats a query as a single, self-contained reasoning problem. A team at Carnegie Mellon University points out that this framing breaks down once a model is having a multi-turn conversation, since a fixed per-turn budget spent generously on an easy opening question leaves less room for a much harder follow-up question three turns later ",[352,430,431],{},[355,432,434],{"href":433},"#source-3","[3]",". Their system, called TAB for turn-adaptive budgets, treats the whole conversation as a sequential compute allocation problem, formally a multi-objective decision process where the model has to decide, turn by turn, how much of a shared token budget to spend now versus hold in reserve for whatever comes later in the exchange. Trained with a group-based reinforcement learning method, TAB was reported to save up to 35 percent of tokens compared with static, turn-unaware budget policies while maintaining comparable accuracy on mathematical reasoning benchmarks, and a variant given advance knowledge of all the sub-questions in a conversation saved up to 40 percent ",[352,437,438],{},[355,439,434],{"href":433},[319,441,340,442,340,446,340,450],{"style":391},[342,443],{"src":444,"alt":445,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1769708115034-9bbe58bad782?w=1200&auto=format&fit=crop","Athletes clearing hurdles of a track and field race, some strides open and easy between barriers and others requiring a full jump, analogous to a multi-turn conversation where some turns are an easy stretch of track and others demand the model's full effort",[397,447,449],{"style":399,"id":448},"pacing-across-an-uneven-track","Pacing Across an Uneven Track",[348,451,452],{"style":404},"A hurdles race is not run at one constant effort. The open stretches between barriers call for a different pace than the barrier itself, and a runner who sprints flat out over every meter arrives at the tenth hurdle with nothing left to clear it. TAB spends its tokens the way that runner spends stride length, freely across the open stretches of a conversation, held back for whichever turn turns out to be the barrier.",[331,454,456],{"id":455},"teaching-the-habit-during-training-not-just-at-inference","Teaching the Habit During Training, Not Just at Inference",[348,458,459,460,466,467,471],{},"A fourth angle on the same problem moves the intervention earlier, into training itself. A framework called BACR, for budget-adaptive curriculum reasoning, argues that training a model under a single fixed token budget, or under budgets sampled uniformly at random, teaches it a compute habit that does not match the actual spread of problem difficulty it will face later ",[352,461,462],{},[355,463,465],{"href":464},"#source-4","[4]",". BACR conditions the model on its assigned budget as an explicit input during training, then uses a scheduler that shifts the distribution of training budgets from easy to hard as the model's own performance improves, paired with a reward that gives partial credit for intermediate reasoning steps even when a response gets cut off by the budget. On a set of mathematics benchmarks including MATH and AIME, the reported result was up to an 8.3 percent accuracy improvement under tight budgets alongside a 34 percent reduction in average token consumption compared with training under unconstrained reasoning ",[352,468,469],{},[355,470,465],{"href":464},". The gain under tight budgets specifically is the detail worth sitting with, since a curriculum that only helps when compute is already generous would be a much smaller contribution.",[331,473,475],{"id":474},"what-this-suggests","What This Suggests",[348,477,478],{},"None of these four systems agree on exactly where the compute-allocation decision should live. Bet puts it inside a single query, TAB puts it across the turns of a conversation, and BACR puts it back in training before any of that decision-making happens at inference time. Taken together they point toward the same underlying claim, that a uniform thinking budget is a specific and avoidable form of waste rather than a neutral default, and that the waste sometimes curdles into an actual accuracy cost through the flip events the Nanjing University team documented. Every reported gain here comes from mathematical reasoning benchmarks or benchmarks adjacent to them, and it remains an open question whether solvability estimates, turn-level budgeting, and curriculum scheduling generalize as cleanly to open-ended tasks like coding or long document analysis, where correctness is harder to check and a flip event is harder to even detect. From a systems perspective, the more durable lesson may be procedural rather than architectural, that any team deploying test-time compute at scale should be measuring the marginal accuracy per token spent, not just the accuracy at whatever budget happens to be the current default.",[319,480,323,484,323,487],{"className":481},[482,483],"references","mt-8",[331,485,486],{"id":482},"References",[488,489,340,495,340,513,340,523,340,533,323],"ol",{"className":490},[491,492,493,494],"list-decimal","list-inside","space-y-2","mt-4",[496,497,499,500,504,505],"li",{"id":498},"source-1","S. Zhou et al., \"When More Thinking Hurts: Overthinking in LLM Test-Time Compute Scaling,\" ",[501,502,503],"em",{},"arXiv",", 2026, ",[355,506,512],{"href":507,"target":508,"className":509},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.10739","_blank",[510,511],"text-blue-600","underline","[Online]",[496,514,516,517,504,519],{"id":515},"source-2","Z. Zhou et al., \"Nice Fold or Hero Call: Learning Budget-Efficient Thinking for Adaptive Reasoning,\" ",[501,518,503],{},[355,520,512],{"href":521,"target":508,"className":522},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.11625",[510,511],[496,524,526,527,504,529],{"id":525},"source-3","N. Jali et al., \"Not All Turns Are Equally Hard: Adaptive Thinking Budgets for Efficient Multi-Turn Reasoning,\" ",[501,528,503],{},[355,530,512],{"href":531,"target":508,"className":532},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.05164",[510,511],[496,534,536,537,504,539],{"id":535},"source-4","A. Rahman et al., \"Avoiding Overthinking and Underthinking: Curriculum-Aware Budget Scheduling for LLMs,\" ",[501,538,503],{},[355,540,512],{"href":541,"target":508,"className":542},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.19780",[510,511],{"title":544,"searchDepth":545,"depth":545,"links":546},"",2,[547,548,552,553,556,557,558],{"id":335,"depth":545,"text":336},{"id":365,"depth":545,"text":366,"children":549},[550],{"id":400,"depth":551,"text":401},3,{"id":408,"depth":545,"text":409},{"id":424,"depth":545,"text":425,"children":554},[555],{"id":448,"depth":551,"text":449},{"id":455,"depth":545,"text":456},{"id":474,"depth":545,"text":475},{"id":482,"depth":545,"text":486},"2026-08-02","Reasoning models that think longer usually score higher, until they don't. New research on test-time compute shows extended reasoning can flip correct answers into wrong ones, and a cluster of recent papers proposes budgeting compute by solvability, by turn, and by training curriculum instead of by a fixed token limit.","md",{"src":344},{"authors":564,"badge":570,"source":572},[565],{"avatar":566,"name":568,"to":569},{"src":567},"\u002Fimg\u002Fmark_avatar.png","Mark Williams","#",{"label":571},"Test-Time Compute",{"name":573,"url":574},"Thinkata Research","https:\u002F\u002Fthinkata.com",true,{"title":242,"description":560},"aQTHbDPuptD50mePoROY3wl3C4CbsGKpNPxyRLhsVpY",[579,827],{"id":580,"title":86,"body":581,"date":815,"description":816,"extension":561,"image":817,"meta":818,"navigation":575,"path":87,"seo":825,"stem":88,"__hash__":826,"_path":87},"insights\u002Fnews\u002Finsights\u002Fai-agent-economics.md",{"type":316,"value":582,"toc":800},[583,595,601,604,612,616,634,647,655,659,671,684,697,701,718,722,735,748,750,753,756],[319,584,323,586,323,590],{"className":585},[322],[325,587,86],{"className":588,"id":589},[328],"winning-the-bid-is-not-the-same-as-knowing-the-price",[331,591,594],{"className":592,"id":593},[334],"a-new-wave-of-auction-benchmarks-keeps-finding-the-same-gap-behind-a-different-door-each-time","A New Wave of Auction Benchmarks Keeps Finding the Same Gap Behind a Different Door Each Time",[319,596,340,597],{"style":339},[342,598],{"src":599,"alt":600,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1702562547661-972d423fe806?w=1200&h=750&fit=crop&crop=focalpoint&fp-x=0.5&fp-y=0.45&auto=format","A woman at a live auction raising her numbered bidding paddle, analogous to how committing to a bid is meant to force an honest, comparable number out of a bidder",[348,602,603],{},"An auction is, underneath the paddle-raising and the countdown, a trick for extracting an honest number out of someone who might not otherwise have one. Ask a bidder in the abstract what a painting is worth and the honest answer is often a shrug. Put the same bidder in a room with rivals and a clock running, and a number gets committed to, backed by real money, which is why economists have leaned on auctions for well over a century as a way of discovering a price that nobody could have simply stated in advance. The number that wins is supposed to mean something. It is supposed to be an honest reflection of what the thing was worth to the person who bid it.",[348,605,606,607,611],{},"Payment networks and retailers are now building toward a version of this where the bidder is a large language model, a system trained to generate text by predicting the next word from patterns in its training data. Anthropic's Project Vend put one in charge of an actual small inventory, and Microsoft's Magentic Marketplace and Alibaba's Shopping Companion point at something similar from the retail side ",[352,608,609],{},[355,610,375],{"href":374},". A handful of benchmarks published in the last few weeks put agents into exactly this kind of room, not to see whether they can complete a purchase, but to see whether the number an agent commits to tracks anything real. What keeps turning up, benchmark after benchmark, is a gap between the willingness to compete for something and an actual grip on what that something is worth. The costume changes. The gap does not.",[331,613,615],{"id":614},"what-winning-does-not-guarantee","What Winning Does Not Guarantee",[348,617,618,619,623,624,628,629,633],{},"A sealed-bid auction, the kind where every bidder writes down an offer without seeing anyone else's, is meant to reward whoever has the clearest read on value, not whoever is most eager to win. One recent benchmark has an LLM merchant bid this way against rival sellers for customers with hidden preferences, and it tracks two very different things: whether the agent won the customer, and how much profit it kept once it had ",[352,620,621],{},[355,622,375],{"href":374},". Across eleven frontier models, those two numbers barely moved together. One model won 10 percentage points more often than its closest competitor and still finished with less money in hand, because winning a customer while charging too little is a specific, quiet way of being wrong about value that a simple leaderboard would never surface ",[352,625,626],{},[355,627,375],{"href":374},". Giving the same models more time to reason before bidding did not close this gap so much as move it. One model's cumulative earnings rose more than sevenfold once allowed to think longer, but the character of its remaining mistakes flipped, trading a habit of losing auctions outright for a habit of winning them too cheaply ",[352,630,631],{},[355,632,375],{"href":374},". More reasoning did not make the number more accurate. It just changed which way the number was wrong.",[319,635,340,636,340,640,340,644],{"style":391},[342,637],{"src":638,"alt":639,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1775948465753-38c8ff55104f?w=1200&auto=format&fit=crop","A crowd watching a car being presented at a live auction, analogous to how winning a bid and knowing the true value of what was won can be two separate things that only careful scoring can tell apart",[397,641,643],{"style":399,"id":642},"the-gavel-coming-down-is-not-the-whole-story","The Gavel Coming Down Is Not the Whole Story",[348,645,646],{"style":404},"A room full of spectators can watch every bid at a live auction and still not know, from the gavel alone, whether the winner got a good deal. Telling the two apart takes a second kind of scorekeeping, one that tracks what was actually captured rather than just who raised a hand first.",[348,648,649,650,654],{},"The same benchmark changed the customers' preferences partway through the run without warning, and the models that had adjusted to those preferences fastest were often the slowest to notice anything had changed. Reading the agents' own written reasoning, researchers found the holdup was rarely a failure to see the loss coming. It was a reluctance to revise a number once it had already been treated as settled, closer to a bidder who keeps returning to the same figure out of habit than one who is still tracking the market ",[352,651,652],{},[355,653,375],{"href":374},". An auction can force a number out of a bidder once. Whether that bidder keeps that number honest as the world underneath it shifts turns out to be a separate question entirely.",[331,656,658],{"id":657},"the-same-gap-multiplied-across-a-room-of-bidders","The Same Gap, Multiplied Across a Room of Bidders",[348,660,661,662,666,667,359],{},"A pricing war is, in effect, a continuous auction that never closes, and a benchmark from Princeton put five LLM-controlled sellers through exactly that, competing for the same customers day after day for a full simulated year ",[352,663,664],{},[355,665,358],{"href":357},". Two of the models tested undercut each other so aggressively that most of the sellers went bankrupt, at which point the lone survivor jacked prices back up to nearly four times cost, a familiar boom-and-bust shape appearing with nobody designing it in. A third model, given the identical rules and the identical rivals, settled into a stable, modestly profitable equilibrium without any coaxing at all. Nothing about the market forced either outcome. Each seller's undercutting looked locally sensible in the moment it happened, cut the price a little, keep today's sale, and the collapse was simply what that same logic adds up to once every seller in the room is doing it ",[352,668,669],{},[355,670,358],{"href":357},[348,672,673,674,678,679,683],{},"The same paper ran a second version of this idea where the currency being bid over was trust rather than price. A used-car marketplace let one deceptive operator run several seller identities at once, a tactic known as a Sybil attack, retiring any identity whose reputation had decayed too far and starting a fresh one under a new name. Buyer agents largely failed to notice the pattern sitting in plain sight in reputation scores, letting a fraudulent operator capture up to 17 percent of the market once nine of twelve sellers were fakes ",[352,675,676],{},[355,677,358],{"href":357},". This is the identical gap from the pricing war, just wearing reputation instead of a dollar figure. Extending trust to a listing is itself a kind of bid, a claim about what the seller's word is worth, and the buyers kept extending it past the point the evidence justified. What is worth noting is which fix actually held up as conditions got harder. Instructing the agents to hold a price floor or double-check a seller's history worked reasonably well and then degraded, while training a much smaller model with reinforcement learning, an approach where a system learns through trial and error rather than through instructions alone, kept its footing under the same pressure that broke the instructed version ",[352,680,681],{},[355,682,358],{"href":357},". Instructions bent. Training held.",[319,685,340,686,340,690,340,694],{"style":391},[342,687],{"src":688,"alt":689,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1779962338482-4c2e90885e05?w=1200&auto=format&fit=crop","A closed storefront with its roller shutter pulled down, analogous to a wave of bankruptcies that followed from each seller's individually reasonable undercutting decision rather than from any single bad actor",[397,691,693],{"style":399,"id":692},"nobody-meant-to-close-the-whole-street","Nobody Meant to Close the Whole Street",[348,695,696],{"style":404},"A shuttered storefront is the visible aftermath of a price war that no single seller necessarily intended, each undercut looked defensible in the moment it was made. Multiply one bidder's uncertain grip on value by an entire street of them, and the mistake stops being a rounding error and starts being the whole market's condition.",[331,698,700],{"id":699},"the-first-bid-decides-who-gets-to-keep-bidding","The First Bid Decides Who Gets to Keep Bidding",[348,702,703,704,708,709,713,714,359],{},"Some of the most consequential bidding in these benchmarks never touches a retail price at all. A supply-chain benchmark from Shanghai Jiao Tong University has twenty retailer agents bid for scarce inventory before setting a price for customers, and that first, unglamorous auction turned out to gate almost everything that followed ",[352,705,706],{},[355,707,434],{"href":433},". Winning early inventory meant having the working capital to bid competitively again later, and losing it meant staying cash-starved for the rest of the run, a compounding advantage that mattered more to profit than pricing skill or marketing ever did. A 14 billion parameter model ended up matching much larger, far more expensive systems on profit, while one model failed to submit a single valid bid across every run and finished at exactly zero ",[352,710,711],{},[355,712,434],{"href":433},". Zero is not a rounding error there. It is exclusion from the game. Winning that first auction for scarce supply was not really a bid on a price. It was a bid on the right to keep playing at all, and the researchers found that the persuasive marketing language agents wrote afterward, though it was directly wired into whether a customer would even see an offer, mattered far less to profit than that initial scramble for inventory, with agents converging on similar, largely interchangeable slogans rather than differentiating once competition set in ",[352,715,716],{},[355,717,434],{"href":433},[331,719,721],{"id":720},"not-every-auction-comes-with-a-price-tag","Not Every Auction Comes With a Price Tag",[348,723,724,725,729,730,734],{},"Strip away price tags and money entirely, and the same shape of finding turns up in a vote. A study of six identical agents, given no assigned economic roles at all, found that an election produced almost no behavioral consequence when the office was ceremonial, work still got distributed by the system regardless of who won, but produced measurable concentration of resources once the same office carried real power to assign scarce daily tasks ",[352,726,727],{},[355,728,465],{"href":464},". Holding the ballot exactly fixed and changing only what winning it actually unlocked was enough to turn an empty ritual into a contest worth having. The same researchers then stripped away the agents' survival stakes entirely, making their resource balances visible but no longer capable of ending anyone's participation, and found that competition for access did not go away. It simply stopped being about staying alive and started being about controlling whatever scarce lever was still real, task promises and vote-trading persisting even after the original prize was gone ",[352,731,732],{},[355,733,465],{"href":464},". The stakes changed. The competing did not. Whatever these agents were bidding for, price tag or ballot or reputation, the thing that decided the outcome was never the label attached to the contest. It was whether anything real sat on the other side of winning.",[319,736,340,737,340,741,340,745],{"style":391},[342,738],{"src":739,"alt":740,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1726499637922-9f544cbc0fd2?w=1200&auto=format&fit=crop","A hand placing a folded ballot into a voting box, analogous to how an identical vote produced very different agent behavior depending on whether the office actually carried authority over something scarce",[397,742,744],{"style":399,"id":743},"the-same-ballot-different-stakes","The Same Ballot, Different Stakes",[348,746,747],{"style":404},"Two elections can look identical on the ballot and still mean entirely different things depending on what the winner is allowed to do afterward. What changed these agents' behavior was never the ceremony of the vote. It was whether the office came with a lever attached to something scarce.",[331,749,475],{"id":474},[348,751,752],{},"Change what is being bid on, a customer's business, a day's pricing, a slot of scarce inventory, a seat with real authority, and the shape of the finding holds steady across all of it. Being willing to compete for something and having an accurate grip on what that something is worth are not obviously the same trait in an AI agent, even though a standard evaluation, and a standard leaderboard, mostly watches the competing and infers the rest. That gap shows up quietly in a single bidder's margin. It multiplies into bankruptcy or fraud once enough bidders are making the same miscalculation in the same room, and it resurfaces again wherever winning unlocks access rather than a price, since access compounds in ways a one-shot bid never fully reveals up front.",[348,754,755],{},"Whether this gap closes as models get better at reasoning about their own incentives, or whether it is a more structural feature of a system trained to produce plausible text rather than to hold a stable, defensible number in mind, is not yet settled by any of this work. What does seem worth taking seriously, for anyone actually wiring agents into markets rather than just chatbots, is that instructions telling an agent to hold a price floor or double-check a seller held up right until conditions got difficult, while training the same behavior into the model directly did not. An auction can still force a number out of an agent. Whether that number can be trusted the way it has traditionally been trusted from a human bidder is the open question these benchmarks were built to start asking.",[319,757,323,759,323,761],{"className":758},[482,483],[331,760,486],{"id":482},[488,762,340,764,340,773,340,782,340,791,323],{"className":763},[491,492,493,494],[496,765,766,767,504,769],{"id":498},"S. Ahmed et al., \"Can LLM Agents Price Competitively? A Dynamic Multi-Attribute Auction Benchmark for Agentic Commerce,\" ",[501,768,503],{},[355,770,512],{"href":771,"target":508,"className":772},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.00102",[510,511],[496,774,775,776,504,778],{"id":515},"S. Karten et al., \"Agent Bazaar: Enabling Economic Alignment in Multi-Agent Marketplaces,\" ",[501,777,503],{},[355,779,512],{"href":780,"target":508,"className":781},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.17698",[510,511],[496,783,784,785,504,787],{"id":525},"Y. Zheng et al., \"Market-Bench: Benchmarking Large Language Models on Economic and Trade Competition,\" ",[501,786,503],{},[355,788,512],{"href":789,"target":508,"className":790},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.05523",[510,511],[496,792,793,794,504,796],{"id":535},"L. Zhang and S. Shang, \"AI Agent Economics: Can Autonomous Economic Behavior Emerge among AI Agents under Minimal External Conditions?,\" ",[501,795,503],{},[355,797,512],{"href":798,"target":508,"className":799},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.03076",[510,511],{"title":544,"searchDepth":545,"depth":545,"links":801},[802,803,806,809,810,813,814],{"id":593,"depth":545,"text":594},{"id":614,"depth":545,"text":615,"children":804},[805],{"id":642,"depth":551,"text":643},{"id":657,"depth":545,"text":658,"children":807},[808],{"id":692,"depth":551,"text":693},{"id":699,"depth":545,"text":700},{"id":720,"depth":545,"text":721,"children":811},[812],{"id":743,"depth":551,"text":744},{"id":474,"depth":545,"text":475},{"id":482,"depth":545,"text":486},"2026-08-09","An auction is supposed to force an honest number out of a bidder who might not otherwise have one, which is why economists have trusted it as a way of discovering value for over a century. New benchmarks that drop AI agents into auctions, pricing wars, and votes over scarce resources keep finding that winning and knowing what the win was actually worth are turning out to be two different skills.",{"src":599},{"authors":819,"badge":822,"source":824},[820],{"avatar":821,"name":568,"to":569},{"src":567},{"label":823},"Agentic Commerce",{"name":573,"url":574},{"title":86,"description":816},"RPqE50hG6n8PjbX4Ji609eO2_UYj0EZHSuQDWg73AfM",{"id":828,"title":74,"body":829,"date":1037,"description":1038,"extension":561,"image":1039,"meta":1040,"navigation":575,"path":75,"seo":1047,"stem":76,"__hash__":1048,"_path":75},"insights\u002Fnews\u002Finsights\u002Fadaptive-context-memory.md",{"type":316,"value":830,"toc":1025},[831,843,849,852,859,863,871,878,891,899,903,911,919,932,936,949,963,965,968],[319,832,323,834,323,838],{"className":833},[322],[325,835,74],{"className":836,"id":837},[328],"nothing-gets-deleted-just-blurred",[331,839,842],{"className":840,"id":841},[334],"long-context-models-are-learning-when-to-zoom-back-in","Long-Context Models Are Learning When to Zoom Back In",[319,844,340,845],{"style":339},[342,846],{"src":847,"alt":848,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1623475329493-889804e377f8?w=1200&auto=format&fit=crop","A curling strip of 35mm slide film with visible frames of buildings and landscapes across its length, analogous to how long-context inference now keeps a compressed thumbnail of most of a conversation rather than deleting it outright",[348,850,851],{},"A strip of developed film holds every frame a photographer shot, but only as a thumbnail. Nothing on the strip is discarded, and nothing on it is sharp enough to read without a loupe or an enlarger. The decision about which frame gets printed at full size happens later, once someone knows which shot actually matters. Large language models handling long documents or long conversations face a version of the same choice, and for the last few years, most systems have chosen the opposite path, throwing frames away entirely rather than shrinking them.",[348,853,854,855,359],{},"The technical name for the film strip inside a language model is the key-value cache, commonly shortened to KV cache. Every time a transformer, the neural network architecture behind most modern language models, processes a token, it computes a key and a value vector for that token in every attention layer, the mechanism that lets the model weigh how much each earlier token matters to the one being generated now. Those vectors get stored so future tokens can attend back to them without recomputing them from scratch. Research from Google on scaling transformer inference showed early on that this cache grows linearly with how much text the model has read, and that the resulting memory pressure, not raw compute, is often what limits how long a context a deployed model can actually afford to hold ",[352,856,857],{},[355,858,375],{"href":374},[331,860,862],{"id":861},"deciding-once-and-living-with-it","Deciding Once and Living With It",[348,864,865,866,870],{},"Most existing fixes to this problem make an early, permanent call. Token eviction methods watch which tokens receive the least attention early on and drop them from the cache for good. Semantic compression methods group tokens into chunks during the initial read and replace each chunk with a single averaged summary before generation even starts. Both approaches work, and both share the same limitation, according to a team from the University of British Columbia and Microsoft Research studying long-context KV caching. A decision about what to keep gets made once, at the beginning, before the model has any idea which later question will actually depend on which earlier passage ",[352,867,868],{},[355,869,358],{"href":357},". Evidence that looked irrelevant in the first paragraph of a legal filing or a codebase might turn out to be the exact clause a later question hinges on, and once it has been evicted or blended into an average, there is no way back.",[348,872,873,874,359],{},"The researchers propose something they call SeKV, short for semantic KV cache, which keeps every span of text in two forms at once rather than picking one. A lightweight summary vector for each span stays resident on the GPU, cheap to scan and used only to decide whether that span looks relevant to the current decoding step. A separate, more detailed representation of the same span, built from a technique called singular value decomposition that factors a block of numbers into a smaller set of components capturing most of its structure, sits on the CPU instead, waiting to be fetched only if the summary suggests it is worth the trip. Segment boundaries themselves come from a signal already available for free during the model's first pass over the text, a measure of how surprising each token is given what came before it, which tends to spike at genuine topic shifts and drop within a coherent run of text ",[352,875,876],{},[355,877,358],{"href":357},[319,879,340,880,340,884,340,888],{"style":391},[342,881],{"src":882,"alt":883,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1722080767146-0ae3a8ff46a0?w=1200&auto=format&fit=crop","A satellite view of a dark lake surrounded by mountains, showing broad shapes clearly but withholding finer ground-level detail until the view is zoomed in, analogous to a coarse memory summary that a model can request higher resolution from on demand",[397,885,887],{"style":399,"id":886},"a-coarse-view-that-can-be-zoomed","A Coarse View That Can Be Zoomed",[348,889,890],{"style":404},"A satellite photograph like this one shows enough to locate a shoreline or a mountain range, but not enough to make out a building or a road. Getting that level of detail means requesting a closer pass over one particular area, not over the whole frame. SeKV applies the same logic to context, routing over cheap summaries first and only reconstructing full detail for the handful of spans a given decoding step actually needs.",[348,892,893,894,898],{},"Whether a given span deserves that closer look gets decided by a small trained component, referred to as a zoom-in mechanism, that scores each summary against the model's current query and expands only the spans that clear a learned threshold. Across four long-context benchmarks, the reported result was a 5.9 percent average improvement over the strongest semantic-compression baseline tested, alongside a 53.3 percent reduction in GPU memory compared with keeping the full cache at a 128,000-token context length, achieved while adding fewer than 0.05 percent additional trainable parameters to a frozen base model ",[352,895,896],{},[355,897,358],{"href":357},". The efficiency gain and the accuracy gain showing up together is worth noting mainly because compression methods usually have to trade one for the other.",[331,900,902],{"id":901},"a-token-that-asks-for-a-refresh","A Token That Asks for a Refresh",[348,904,905,906,910],{},"A separate team, working across several Chinese research labs, tackled a closely related question from another angle. Rather than deciding what to keep once and reconstructing detail on demand, their system, called PReM for preserve and refresh memory, lets the model itself decide when its current compressed view of the context has gone stale and needs to be rebuilt ",[352,907,908],{},[355,909,434],{"href":433},". PReM trains a special memory token that the model can emit mid-generation. Emitting it triggers a fresh look back over the full context, re-selecting which chunks deserve to be kept at full token-level detail and which get replaced with an averaged stand-in, based on whatever the model is trying to figure out at that specific step rather than on whatever seemed important when the context was first read.",[348,912,913,914,918],{},"Training a model to condition its own generation on a memory state that changes partway through, rather than staying fixed for the whole response, required the researchers to split each generation step into two separate forward passes, one that handles the refresh decision and one that generates conditioned on the result. On a set of question-answering benchmarks using 32,000-token contexts, PReM reportedly outperformed eight established KV-cache and context-compression baselines under both 16 times and 32 times compression, improving average exact-match and F1 scores by up to 10.23 and 12.55 points respectively over the strongest baseline tested ",[352,915,916],{},[355,917,434],{"href":433},". A smaller 3-billion-parameter version of the model was also reported to outperform some larger 7-billion-parameter compression baselines, which the authors take as a sign that learning when to refresh can partly substitute for raw model scale, at least on the benchmarks tried so far.",[319,920,340,921,340,925,340,929],{"style":391},[342,922],{"src":923,"alt":924,"style":346},"https:\u002F\u002Fimages.unsplash.com\u002Fphoto-1573166364266-356ef04ae798?w=1200&auto=format&fit=crop","A person writing on a dry-erase whiteboard, part of the surface freshly wiped clean while other notes remain, analogous to a memory token that triggers a partial refresh of what a model keeps in view without discarding the underlying source material",[397,926,928],{"style":399,"id":927},"refreshing-the-board-without-losing-the-room","Refreshing the Board Without Losing the Room",[348,930,931],{"style":404},"Wiping part of a whiteboard clean does not erase the meeting that produced the notes, only the summary written down at the time. PReM's memory token works on a similar principle, refreshing the compressed view the model is currently attending to while the full source context stays available underneath, ready to be resummarized differently the next time a refresh is triggered.",[331,933,935],{"id":934},"memory-that-outlives-a-single-document","Memory That Outlives a Single Document",[348,937,938,939,943,944,948],{},"Both SeKV and PReM operate within a single long document or a single extended answer. A different line of work asks what changes when the relevant memory spans many separate conversations with the same user over time, a setting closer to how a deployed assistant actually gets used. A team centered at the University of Electronic Science and Technology of China proposes LightMem, which splits an agent's memory into three tiers, a short-term store for the immediate conversation, a mid-term store of reusable summaries from recent sessions, and a long-term store of consolidated knowledge, each managed by a small, purpose-built language model rather than a large one ",[352,940,941],{},[355,942,465],{"href":464},". Retrieval happens in two cheap stages, a coarse vector search followed by a semantic consistency check, before anything gets handed to the main model. Reported results on a long-term dialogue benchmark showed roughly a 2.5-point average gain in F1 score across model scales, with a median end-to-end latency around 581 milliseconds and retrieval alone completing in about 83 milliseconds ",[352,945,946],{},[355,947,465],{"href":464},". The appeal of the approach, from a systems perspective, sits less in the accuracy number and more in the latency figure, since memory operations that run on small models rather than the main large model avoid competing for the same expensive compute.",[348,950,951,952,958,959,359],{},"A related but distinct idea comes from a team at the University of Science and Technology of China, who point out that agents which compress a long document into memory through a single linear pass, reading start to finish and updating a running summary as they go, tend to prune evidence early that only turns out to matter once a much later part of the document has been read ",[352,953,954],{},[355,955,957],{"href":956},"#source-5","[5]",". Their system, ReMemR1, lets an agent issue what the authors call a callback query mid-reasoning, retrieving an earlier memory state instead of only ever moving forward through it, and trains the behavior with a reward signal that combines the final answer's correctness with denser, step-level feedback about whether a given callback was actually useful. On long-context question-answering benchmarks, the reported result was upward of a 20 percent relative reduction in error rate compared with linear memory baselines, with the retrieval mechanism adding less than 0.2 percent to overall computation time ",[352,960,961],{},[355,962,957],{"href":956},[331,964,475],{"id":474},[348,966,967],{},"Four independent groups converging on some version of the same idea, that compression decisions should stay reversible and resolution should be allocated on demand rather than fixed up front, is a pattern worth watching rather than a settled conclusion. Every result above comes from a specific benchmark and a specific context length, 32,000 tokens for PReM, 128,000 for SeKV, documents padded to a few thousand entries for ReMemR1, and none of them yet speaks directly to million-token agent contexts or to workloads where the CPU-to-GPU transfer that SeKV and similar systems depend on becomes a bottleneck of its own under real production load. Whether the accuracy gains reported here persist once these methods leave curated benchmarks and meet the messier, longer-horizon contexts that production agents actually accumulate remains an open question, and one that will likely need answering system by system rather than in the abstract.",[319,969,323,971,323,973],{"className":970},[482,483],[331,972,486],{"id":482},[488,974,340,976,340,987,340,996,340,1005,340,1014,323],{"className":975},[491,492,493,494],[496,977,978,979,982,983],{"id":498},"R. Pope et al., \"Efficiently Scaling Transformer Inference,\" ",[501,980,981],{},"Proceedings of the Sixth Conference on Machine Learning and Systems (MLSys 2023)",", 2023. DOI: ",[355,984,512],{"href":985,"target":508,"className":986},"https:\u002F\u002Fdoi.org\u002F10.48550\u002FarXiv.2211.05102",[510,511],[496,988,989,990,504,992],{"id":515},"A. Abaskohi et al., \"SeKV: Resolution-Adaptive KV Cache with Hierarchical Semantic Memory for Long-Context LLM Inference,\" ",[501,991,503],{},[355,993,512],{"href":994,"target":508,"className":995},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.31145",[510,511],[496,997,998,999,504,1001],{"id":525},"B. Yu et al., \"PReM: Learning What to Preserve and When to Refresh for Context Compression,\" ",[501,1000,503],{},[355,1002,512],{"href":1003,"target":508,"className":1004},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2607.14327",[510,511],[496,1006,1007,1008,504,1010],{"id":535},"J. Zhang et al., \"Lightweight LLM Agent Memory with Small Language Models,\" ",[501,1009,503],{},[355,1011,512],{"href":1012,"target":508,"className":1013},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.07798",[510,511],[496,1015,1017,1018,1020,1021],{"id":1016},"source-5","Y. Shi et al., \"Look Back to Reason Forward: Revisitable Memory for Long-Context LLM Agents,\" ",[501,1019,503],{},", 2025, ",[355,1022,512],{"href":1023,"target":508,"className":1024},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.23040",[510,511],{"title":544,"searchDepth":545,"depth":545,"links":1026},[1027,1028,1031,1034,1035,1036],{"id":841,"depth":545,"text":842},{"id":861,"depth":545,"text":862,"children":1029},[1030],{"id":886,"depth":551,"text":887},{"id":901,"depth":545,"text":902,"children":1032},[1033],{"id":927,"depth":551,"text":928},{"id":934,"depth":545,"text":935},{"id":474,"depth":545,"text":475},{"id":482,"depth":545,"text":486},"2026-07-27","Long-context language models have relied on discarding or freezing most of what they read to keep memory costs down. New research on resolution-adaptive caching and refreshable memory tokens suggests a different approach, keeping everything at low resolution and reconstructing detail only when a later step actually needs it.",{"src":847},{"authors":1041,"badge":1044,"source":1046},[1042],{"avatar":1043,"name":568,"to":569},{"src":567},{"label":1045},"Long-Context Inference",{"name":573,"url":574},{"title":74,"description":1038},"JZHObIsLZXHYqMm_rMP1pKlqCXS6qzRMaqUmKazEsmQ",1787055018842]