1
/**2
* Default one-shot summarization and durable checkpoint framing.3
*4
* @module @deepseek-ai/dsh-compaction-basic/summarizer5
*/7
import type { Context } from '@deepseek-ai/cordis'8
import { contentHasImage, BlockAssembler, LlmError } from '@deepseek-ai/dsh-llm'9
import { deepFreeze } from '@deepseek-ai/dsh-util-values'10
import type {11
ContentBlock, FinishReason, GenerateOptions, Message, RequestMessage, TokenUsage, ToolSchema,12
} from '@deepseek-ai/dsh-llm'13
import type { Agent } from '@deepseek-ai/dsh-agent'15
interface SummaryConfig {16
readonly summarizationProvider: string17
readonly summarizationModel: string18
readonly maxTokens: number19
}21
/** Tags wrapping the structured summary inside the landed checkpoint node. */22
const SUMMARY_OPEN_TAG = '<compacted-summary>'23
const SUMMARY_CLOSE_TAG = '</compacted-summary>'25
/**26
* The summarization directive, delivered as the FINAL user message after the27
* replayed conversation rather than as a distinct summarizer system prompt.28
* Keeping the conversation's own system prompt, tools, and message prefix in29
* front of it makes the auxiliary call a genuine prefix of the last routed30
* request, so the provider's KV cache is reused instead of invalidated.31
*/32
const COMPACTION_INSTRUCTION = [33
'You are now acting as a compaction engine for this AI coding assistant. Condense the conversation ABOVE into a structured checkpoint that lets another model resume the work with no loss of essential context.',34
'',35
'Output EXACTLY the Markdown structure below: keep every section, in order. Use terse bullets, not prose paragraphs. Write "(none)" for an empty section — never drop a section.',36
'',37
'## Primary Request and Intent',38
"- [the user's original and evolving goals; quote verbatim where the exact wording matters]",39
'',40
'## Key Technical Concepts',41
'- [technologies, frameworks, patterns, and conventions in play]',42
'',43
'## Files and Code',44
'- [exact path: why it matters, key changes or snippets]',45
'',46
'## Errors and Fixes',47
'- [error: how it was resolved, plus any related user feedback]',48
'',49
'## Pending Jobs',50
'- [explicitly requested work not yet completed]',51
'',52
'## Current Work',53
'- [precisely what was in progress at this checkpoint]',54
'',55
'## Next Step',56
'- [the single next action, directly in line with the most recent request, or "(none)"]',57
'',58
'## Critical Context',59
'- [decisions and their rationale, constraints, user preferences, open questions, data needed to continue]',60
'',61
'Rules:',62
'- Write concise English engineering prose. Preserve exact file paths, commands, error strings, identifiers, numeric values, function signatures, and syntax fragments.',63
'- Capture user feedback and explicit instructions faithfully, especially corrections.',64
'- Do NOT mention this summarization request or that the context was compacted.',65
'- Output only the checkpoint text: do not call any tool or take any other action.',66
`- If the conversation already contains a ${SUMMARY_OPEN_TAG} block, it is a PRIOR checkpoint. Do not copy it forward verbatim: preserve still-true facts, drop stale ones, and merge newer information into a single consolidated summary under the same structure.`,67
].join('\n')69
/** Framing that makes the replacement user message established context. */70
const CHECKPOINT_PREAMBLE =71
'This is an automatically generated checkpoint condensing an earlier span of the conversation to free up context. Treat the captured context as established background and build on it without restating it. Continue the task directly from the messages that follow, without acknowledging this checkpoint.'73
/**74
* The replayed conversation surface the summarizer condenses. Reproducing the75
* last routed request's system prompt, tools, and leading messages verbatim76
* lets the auxiliary call reuse the provider's warm prefix cache; the trailing77
* compaction instruction is then the only novel input.78
*/79
export interface SummarizationInput {80
/** The conversation's tool schemas, reused for prefix-cache alignment; absent when the request carried none. */81
readonly tools?: readonly ToolSchema[]82
/** The derived system head, when present, followed by the shadowed region in surface order. */83
readonly messages: readonly Message[]84
}86
/** Safe summary content plus the exact auxiliary call envelope recorded with it. */87
export type SummaryResult = {88
summary: ContentBlock[]89
provider: string90
model: string91
maxTokens?: number92
/** Provider-reported usage for this summarization request. */93
usage?: TokenUsage94
} & (95
| {96
/** Complete provider output before the text-only summary projection. */97
rawOutput: ContentBlock[]98
/** Identifies exactly one call through this context's `ctx.llm.stream()`. */99
llmStreamCall: true100
}101
| {102
/** Optional complete output from an unmarked template, remote, or other summarizer. */103
rawOutput?: ContentBlock[]104
/** An unmarked result does not identify a call through this context's LLM seam. */105
llmStreamCall?: never106
}107
)109
/**110
* Run the default cache-reusing `ctx.llm.stream()` summarization call: replay111
* the conversation prefix, then append the compaction instruction as the final112
* user message so the provider's warm prefix cache is reused.113
* @param ctx - context providing the LLM service.114
* @param config - resolved backend configuration.115
* @param input - replayed conversation prefix (system, tools, and leading messages) to condense.116
* @param agent - supplies routed-model history, fallback model, and session id.117
* @param signal - optional cancellation forwarded to the adapter.118
* @returns safe text-only summary blocks and the exact call envelope and output.119
*/120
export async function summarizeWithLlm(121
ctx: Context,122
config: SummaryConfig,123
input: SummarizationInput,124
agent: Agent,125
signal?: AbortSignal,126
): Promise<SummaryResult> {127
const latest = agent.session.requestHeader()?.config128
const configured = config.summarizationProvider.length === 0129
? undefined130
: { provider: config.summarizationProvider, model: config.summarizationModel }131
const agentTarget = agent.options.provider !== undefined132
&& agent.options.provider.length > 0133
&& agent.options.model !== undefined134
&& agent.options.model.length > 0135
? { provider: agent.options.provider, model: agent.options.model }136
: undefined137
const target = configured ?? latest ?? agentTarget138
if (target === undefined) {139
throw new Error(140
'no provider/model available for summarization: set both BasicCompactionConfig summarization fields, route one request, or set both AgentOptions fields',141
)142
}144
const assembler = new BlockAssembler()145
const messages: RequestMessage[] = [146
...input.messages,147
deepFreeze({148
role: 'user',149
content: [{ type: 'text', text: COMPACTION_INSTRUCTION }],150
}),151
]152
const options: GenerateOptions = {153
provider: target.provider,154
model: target.model,155
messages,156
toolHistory: agent.session.toolHistory(),157
...input.tools === undefined ? {} : { tools: [...input.tools] },158
maxTokens: config.maxTokens,159
sessionId: agent.session.id,160
purpose: 'compaction',161
...signal === undefined ? {} : { signal },162
}163
for await (const chunk of ctx.llm.stream(options)) assembler.push(chunk)164
const error = finishError(assembler.finish)165
if (error !== undefined) throw error167
const rawOutput = assembler.blocks()168
const summary = summaryText(rawOutput)169
if (!summary.some(block => block.text.trim().length > 0)) {170
throw new Error('summarization produced no text summary content')171
}172
return {173
summary,174
rawOutput,175
llmStreamCall: true,176
provider: options.provider,177
model: options.model,178
maxTokens: config.maxTokens,179
...(assembler.usage === undefined ? {} : { usage: assembler.usage }),180
}181
}183
/**184
* Wrap raw summary blocks in the durable checkpoint framing.185
* @param summary - safe text-only model output.186
* @returns content for the synthesized replacement user message.187
*/188
export function frameSummary(summary: readonly ContentBlock[]): ContentBlock[] {189
return [190
{ type: 'text', text: `${CHECKPOINT_PREAMBLE}\n\n${SUMMARY_OPEN_TAG}` },191
...summary,192
{ type: 'text', text: SUMMARY_CLOSE_TAG },193
]194
}196
/** Map a terminal summarization finish to its fail-closed error. */197
function finishError(finish: FinishReason): Error | undefined {198
switch (finish.kind) {199
case 'error':200
case 'aborted': {201
return new LlmError(finish.failure.message, finish.failure.code, finish.failure)202
}203
case 'max-tokens': {204
const error = new Error('summarization truncated at the token cap (incomplete checkpoint)') as Error & { code?: string }205
error.code = 'MAX_TOKENS'206
return error207
}208
default:209
return undefined210
}211
}213
/** Reject visual output and keep only text before synthesizing a user message. */214
function summaryText(215
blocks: readonly ContentBlock[],216
): Array<Extract<ContentBlock, { type: 'text' }>> {217
if (contentHasImage(blocks)) {218
throw new LlmError('compaction summary cannot contain image output', 'UNSUPPORTED_CONTENT')219
}220
return blocks.filter((block): block is Extract<ContentBlock, { type: 'text' }> => block.type === 'text')221
}