@@ -98,6 +98,87 @@ export function hasCompactableHistory(messages: Message[]): boolean {
9898 )
9999}
100100
101+ /** The messages a model compaction keeps verbatim after its summary: the
102+ * latest instructions and the live user request (with its steering). */
103+ function compactionSuffix ( messages : Message [ ] ) : Message [ ] {
104+ const lastPrompt = messages . findLastIndex ( ( m ) =>
105+ m . tags ?. includes ( 'USER_PROMPT' ) ,
106+ )
107+ let promptStart = lastPrompt
108+ while (
109+ promptStart > 0 &&
110+ messages [ promptStart - 1 ] . tags ?. includes ( 'USER_PROMPT' )
111+ )
112+ promptStart --
113+ const live =
114+ promptStart < 0
115+ ? [ ]
116+ : messages
117+ . slice ( promptStart )
118+ . filter ( ( m ) => m . tags ?. includes ( 'USER_PROMPT' ) )
119+ const instructions = messages . findLast ( ( m ) =>
120+ m . tags ?. includes ( 'INSTRUCTIONS_PROMPT' ) ,
121+ )
122+ return [ ...( instructions ? [ instructions ] : [ ] ) , ...live ]
123+ }
124+
125+ function compactionSummaryBudget ( params : {
126+ maxContextLength : number
127+ fixedTokenCount : number
128+ suffixTokens : number
129+ } ) : number {
130+ return Math . min (
131+ SUMMARY_LIMIT ,
132+ Math . floor (
133+ ( params . maxContextLength - params . fixedTokenCount - params . suffixTokens ) /
134+ 3 ,
135+ ) ,
136+ )
137+ }
138+
139+ /**
140+ * The share of the trigger threshold a compaction's result may occupy for the
141+ * automatic trigger to be worth firing. Above it, the next tool result or two
142+ * crosses the threshold again and the run compacts its own summary.
143+ */
144+ export const COMPACTION_LOW_WATER = 0.85
145+
146+ /**
147+ * The largest context a model compaction can leave behind: the fixed prefix
148+ * (system prompt, tool schemas), the live request it keeps verbatim, and the
149+ * summary budget it asks for. None of it is compactable, so when this is not
150+ * comfortably under the threshold, compacting at the threshold only buys a
151+ * few thousand tokens before the next one: a 32k BYOK window with ~16k of
152+ * Desktop tool schemas compacted every few tool calls, each pass summarizing
153+ * the last summary.
154+ */
155+ export function compactedContextCeiling ( params : {
156+ messages : Message [ ]
157+ maxContextLength : number
158+ fixedTokenCount : number
159+ } ) : number {
160+ const suffixTokens = countTokensMessages ( compactionSuffix ( params . messages ) )
161+ const summaryBudget = compactionSummaryBudget ( { ...params , suffixTokens } )
162+ return params . fixedTokenCount + suffixTokens + Math . max ( 0 , summaryBudget )
163+ }
164+
165+ /**
166+ * Whether an automatic compaction at `thresholdTokens` leaves real room to
167+ * work in. When it cannot, the run keeps its history until the hard budget,
168+ * where compaction is no longer optional.
169+ */
170+ export function automaticCompactionIsWorthwhile ( params : {
171+ messages : Message [ ]
172+ maxContextLength : number
173+ thresholdTokens : number
174+ fixedTokenCount : number
175+ } ) : boolean {
176+ return (
177+ compactedContextCeiling ( params ) <=
178+ Math . floor ( params . thresholdTokens * COMPACTION_LOW_WATER )
179+ )
180+ }
181+
101182/** A model handoff, not a mechanical reduction of tool results. Nothing mutates
102183 * the source history until every section has a valid, bounded result. */
103184export async function compactWithModel ( params : {
@@ -121,34 +202,12 @@ export async function compactWithModel(params: {
121202 countTokensMessages ( params . messages ) + params . fixedTokenCount
122203 // A compact-only request never enters the history. Keep the actual current
123204 // user request verbatim, including steering and attachments.
124- const lastPrompt = params . messages . findLastIndex ( ( m ) =>
125- m . tags ?. includes ( 'USER_PROMPT' ) ,
126- )
127- let promptStart = lastPrompt
128- while (
129- promptStart > 0 &&
130- params . messages [ promptStart - 1 ] . tags ?. includes ( 'USER_PROMPT' )
131- )
132- promptStart --
133- const live =
134- promptStart < 0
135- ? [ ]
136- : params . messages
137- . slice ( promptStart )
138- . filter ( ( m ) => m . tags ?. includes ( 'USER_PROMPT' ) )
139- const instructions = params . messages . findLast ( ( m ) =>
140- m . tags ?. includes ( 'INSTRUCTIONS_PROMPT' ) ,
141- )
142- const suffix = [ ...( instructions ? [ instructions ] : [ ] ) , ...live ]
143- const summaryBudget = Math . min (
144- SUMMARY_LIMIT ,
145- Math . floor (
146- ( params . maxContextLength -
147- params . fixedTokenCount -
148- countTokensMessages ( suffix ) ) /
149- 3 ,
150- ) ,
151- )
205+ const suffix = compactionSuffix ( params . messages )
206+ const summaryBudget = compactionSummaryBudget ( {
207+ maxContextLength : params . maxContextLength ,
208+ fixedTokenCount : params . fixedTokenCount ,
209+ suffixTokens : countTokensMessages ( suffix ) ,
210+ } )
152211 if ( summaryBudget < 256 )
153212 throw new Error (
154213 'The current request and instructions leave too little room to compact. Shorten the request or configure a larger context window.' ,
@@ -319,6 +378,14 @@ export async function compactWithModelOrFallback(
319378 runId ?: string
320379 model ?: string
321380 trigger ?: string
381+ /**
382+ * Where the mechanical fallback should aim, below `maxContextLength`. That
383+ * pass fills whatever budget it is given, so aimed at the hard budget it
384+ * lands above an automatic trigger's threshold and the very next step
385+ * compacts again. Falls back to `maxContextLength` when the target is too
386+ * small to hold the live request.
387+ */
388+ fallbackTargetTokens ?: number
322389 } ,
323390) : Promise < {
324391 messages : Message [ ]
@@ -327,22 +394,41 @@ export async function compactWithModelOrFallback(
327394 postTokens : number
328395 fallback ?: true
329396} | null > {
330- const { logger, runId, model, trigger, ...modelParams } = params
397+ const {
398+ logger,
399+ runId,
400+ model,
401+ trigger,
402+ fallbackTargetTokens,
403+ ...modelParams
404+ } = params
331405 try {
332406 return await compactWithModel ( modelParams )
333407 } catch ( error ) {
334408 if ( params . signal . aborted || isAbortError ( error ) ) throw error
335409 const errorMessage = error instanceof Error ? error . message : String ( error )
336410 let fallback : ReturnType < typeof compactHistoryNow > = null
337411 let fallbackError : string | undefined
338- try {
339- fallback = compactHistoryNow ( {
412+ const mechanical = ( maxContextLength : number ) =>
413+ compactHistoryNow ( {
340414 messages : params . messages ,
341- maxContextLength : params . maxContextLength ,
415+ maxContextLength,
342416 fixedTokenCount : params . fixedTokenCount ,
343417 logger,
344418 runId,
345419 } )
420+ try {
421+ if (
422+ fallbackTargetTokens !== undefined &&
423+ fallbackTargetTokens < params . maxContextLength
424+ ) {
425+ try {
426+ fallback = mechanical ( fallbackTargetTokens )
427+ } catch {
428+ // The live request does not fit the target; use the whole budget.
429+ }
430+ }
431+ fallback ??= mechanical ( params . maxContextLength )
346432 } catch ( mechanicalError ) {
347433 fallbackError =
348434 mechanicalError instanceof Error
0 commit comments