返回源码地图

packages/compaction/compaction-basic/src/summarizer.ts

main snapshot · da00f7f5358f · 正文引用章节 09;完整原文可核对,不声称全文件人工逐行审计

完整原文供逐行核对;页面收录不代表每行都经过人工语义审核。MIT 许可见 许可证。

1/**
2 * Default one-shot summarization and durable checkpoint framing.
3 *
4 * @module @deepseek-ai/dsh-compaction-basic/summarizer
5 */
6
7import type { Context } from '@deepseek-ai/cordis'
8import { contentHasImage, BlockAssembler, LlmError } from '@deepseek-ai/dsh-llm'
9import { deepFreeze } from '@deepseek-ai/dsh-util-values'
10import type {
11 ContentBlock, FinishReason, GenerateOptions, Message, RequestMessage, TokenUsage, ToolSchema,
12} from '@deepseek-ai/dsh-llm'
13import type { Agent } from '@deepseek-ai/dsh-agent'
14
15interface SummaryConfig {
16 readonly summarizationProvider: string
17 readonly summarizationModel: string
18 readonly maxTokens: number
19}
20
21/** Tags wrapping the structured summary inside the landed checkpoint node. */
22const SUMMARY_OPEN_TAG = '<compacted-summary>'
23const SUMMARY_CLOSE_TAG = '</compacted-summary>'
24
25/**
26 * The summarization directive, delivered as the FINAL user message after the
27 * replayed conversation rather than as a distinct summarizer system prompt.
28 * Keeping the conversation's own system prompt, tools, and message prefix in
29 * front of it makes the auxiliary call a genuine prefix of the last routed
30 * request, so the provider's KV cache is reused instead of invalidated.
31 */
32const COMPACTION_INSTRUCTION = [
33 'You are now acting as a compaction engine for this AI coding assistant. Condense the conversation ABOVE into a structured checkpoint that lets another model resume the work with no loss of essential context.',
34 '',
35 'Output EXACTLY the Markdown structure below: keep every section, in order. Use terse bullets, not prose paragraphs. Write "(none)" for an empty section — never drop a section.',
36 '',
37 '## Primary Request and Intent',
38 "- [the user's original and evolving goals; quote verbatim where the exact wording matters]",
39 '',
40 '## Key Technical Concepts',
41 '- [technologies, frameworks, patterns, and conventions in play]',
42 '',
43 '## Files and Code',
44 '- [exact path: why it matters, key changes or snippets]',
45 '',
46 '## Errors and Fixes',
47 '- [error: how it was resolved, plus any related user feedback]',
48 '',
49 '## Pending Jobs',
50 '- [explicitly requested work not yet completed]',
51 '',
52 '## Current Work',
53 '- [precisely what was in progress at this checkpoint]',
54 '',
55 '## Next Step',
56 '- [the single next action, directly in line with the most recent request, or "(none)"]',
57 '',
58 '## Critical Context',
59 '- [decisions and their rationale, constraints, user preferences, open questions, data needed to continue]',
60 '',
61 'Rules:',
62 '- Write concise English engineering prose. Preserve exact file paths, commands, error strings, identifiers, numeric values, function signatures, and syntax fragments.',
63 '- Capture user feedback and explicit instructions faithfully, especially corrections.',
64 '- Do NOT mention this summarization request or that the context was compacted.',
65 '- Output only the checkpoint text: do not call any tool or take any other action.',
66 `- If the conversation already contains a ${SUMMARY_OPEN_TAG} block, it is a PRIOR checkpoint. Do not copy it forward verbatim: preserve still-true facts, drop stale ones, and merge newer information into a single consolidated summary under the same structure.`,
67].join('\n')
68
69/** Framing that makes the replacement user message established context. */
70const CHECKPOINT_PREAMBLE =
71 'This is an automatically generated checkpoint condensing an earlier span of the conversation to free up context. Treat the captured context as established background and build on it without restating it. Continue the task directly from the messages that follow, without acknowledging this checkpoint.'
72
73/**
74 * The replayed conversation surface the summarizer condenses. Reproducing the
75 * last routed request's system prompt, tools, and leading messages verbatim
76 * lets the auxiliary call reuse the provider's warm prefix cache; the trailing
77 * compaction instruction is then the only novel input.
78 */
79export interface SummarizationInput {
80 /** The conversation's tool schemas, reused for prefix-cache alignment; absent when the request carried none. */
81 readonly tools?: readonly ToolSchema[]
82 /** The derived system head, when present, followed by the shadowed region in surface order. */
83 readonly messages: readonly Message[]
84}
85
86/** Safe summary content plus the exact auxiliary call envelope recorded with it. */
87export type SummaryResult = {
88 summary: ContentBlock[]
89 provider: string
90 model: string
91 maxTokens?: number
92 /** Provider-reported usage for this summarization request. */
93 usage?: TokenUsage
94} & (
95 | {
96 /** Complete provider output before the text-only summary projection. */
97 rawOutput: ContentBlock[]
98 /** Identifies exactly one call through this context's `ctx.llm.stream()`. */
99 llmStreamCall: true
100 }
101 | {
102 /** Optional complete output from an unmarked template, remote, or other summarizer. */
103 rawOutput?: ContentBlock[]
104 /** An unmarked result does not identify a call through this context's LLM seam. */
105 llmStreamCall?: never
106 }
107)
108
109/**
110 * Run the default cache-reusing `ctx.llm.stream()` summarization call: replay
111 * the conversation prefix, then append the compaction instruction as the final
112 * user message so the provider's warm prefix cache is reused.
113 * @param ctx - context providing the LLM service.
114 * @param config - resolved backend configuration.
115 * @param input - replayed conversation prefix (system, tools, and leading messages) to condense.
116 * @param agent - supplies routed-model history, fallback model, and session id.
117 * @param signal - optional cancellation forwarded to the adapter.
118 * @returns safe text-only summary blocks and the exact call envelope and output.
119 */
120export async function summarizeWithLlm(
121 ctx: Context,
122 config: SummaryConfig,
123 input: SummarizationInput,
124 agent: Agent,
125 signal?: AbortSignal,
126): Promise<SummaryResult> {
127 const latest = agent.session.requestHeader()?.config
128 const configured = config.summarizationProvider.length === 0
129 ? undefined
130 : { provider: config.summarizationProvider, model: config.summarizationModel }
131 const agentTarget = agent.options.provider !== undefined
132 && agent.options.provider.length > 0
133 && agent.options.model !== undefined
134 && agent.options.model.length > 0
135 ? { provider: agent.options.provider, model: agent.options.model }
136 : undefined
137 const target = configured ?? latest ?? agentTarget
138 if (target === undefined) {
139 throw new Error(
140 'no provider/model available for summarization: set both BasicCompactionConfig summarization fields, route one request, or set both AgentOptions fields',
141 )
142 }
143
144 const assembler = new BlockAssembler()
145 const messages: RequestMessage[] = [
146 ...input.messages,
147 deepFreeze({
148 role: 'user',
149 content: [{ type: 'text', text: COMPACTION_INSTRUCTION }],
150 }),
151 ]
152 const options: GenerateOptions = {
153 provider: target.provider,
154 model: target.model,
155 messages,
156 toolHistory: agent.session.toolHistory(),
157 ...input.tools === undefined ? {} : { tools: [...input.tools] },
158 maxTokens: config.maxTokens,
159 sessionId: agent.session.id,
160 purpose: 'compaction',
161 ...signal === undefined ? {} : { signal },
162 }
163 for await (const chunk of ctx.llm.stream(options)) assembler.push(chunk)
164 const error = finishError(assembler.finish)
165 if (error !== undefined) throw error
166
167 const rawOutput = assembler.blocks()
168 const summary = summaryText(rawOutput)
169 if (!summary.some(block => block.text.trim().length > 0)) {
170 throw new Error('summarization produced no text summary content')
171 }
172 return {
173 summary,
174 rawOutput,
175 llmStreamCall: true,
176 provider: options.provider,
177 model: options.model,
178 maxTokens: config.maxTokens,
179 ...(assembler.usage === undefined ? {} : { usage: assembler.usage }),
180 }
181}
182
183/**
184 * Wrap raw summary blocks in the durable checkpoint framing.
185 * @param summary - safe text-only model output.
186 * @returns content for the synthesized replacement user message.
187 */
188export function frameSummary(summary: readonly ContentBlock[]): ContentBlock[] {
189 return [
190 { type: 'text', text: `${CHECKPOINT_PREAMBLE}\n\n${SUMMARY_OPEN_TAG}` },
191 ...summary,
192 { type: 'text', text: SUMMARY_CLOSE_TAG },
193 ]
194}
195
196/** Map a terminal summarization finish to its fail-closed error. */
197function finishError(finish: FinishReason): Error | undefined {
198 switch (finish.kind) {
199 case 'error':
200 case 'aborted': {
201 return new LlmError(finish.failure.message, finish.failure.code, finish.failure)
202 }
203 case 'max-tokens': {
204 const error = new Error('summarization truncated at the token cap (incomplete checkpoint)') as Error & { code?: string }
205 error.code = 'MAX_TOKENS'
206 return error
207 }
208 default:
209 return undefined
210 }
211}
212
213/** Reject visual output and keep only text before synthesizing a user message. */
214function summaryText(
215 blocks: readonly ContentBlock[],
216): Array<Extract<ContentBlock, { type: 'text' }>> {
217 if (contentHasImage(blocks)) {
218 throw new LlmError('compaction summary cannot contain image output', 'UNSUPPORTED_CONTENT')
219 }
220 return blocks.filter((block): block is Extract<ContentBlock, { type: 'text' }> => block.type === 'text')
221}