返回 DeepSeek-Reasonix
compact.go
根目录 / internal / agent / compact.go
1 package agent
2
3 import (
4 "context"
5 "encoding/json"
6 "errors"
7 "fmt"
8 "sort"
9 "strings"
10 "unicode/utf8"
11
12 "reasonix/internal/ablation"
13 "reasonix/internal/event"
14 "reasonix/internal/provider"
15 )
16
17 // Compaction is a low-frequency cache-reset point: the prompt grows append-only
18 // until compactRatio of the window is crossed, then pressure-time tool pruning
19 // and up to two cache-aligned summary checkpoints restore headroom.
20 const (
21 defaultCompactRatio = 0.80 // sole automatic maintenance trigger (new configs)
22 recentTailBudgetRatio = 0.16 // recent verbatim tail as a fraction of the window
23 summaryOutputMaxTokens = 8192 // max digest output; further clipped by remaining candidate space
24
25 // summaryReasoningMaxBytes clamps a surfaced reasoning-only summary
26 // (~8k tokens of bytes), matching the summaryOutputMaxTokens envelope.
27 summaryReasoningMaxBytes = 32768
28
29 minRecentKeep = 2 // never keep fewer recent messages than this
30 minCompactMessages = 2 // skip compaction below this many compactable messages
31 fallbackTokPerChar = 0.25 // ~4 chars/token, used before any usage is available to calibrate
32 protocolReserveTokens = 256 // provider framing and control fields not represented by message estimates
33 )
34
35 var (
36 errSummaryOutputTruncated = errors.New("summarizer output truncated")
37 errCheckpointRejected = errors.New("checkpoint candidate rejected")
38 )
39
40 // summaryTag wraps the compaction summary so the model can distinguish it from
41 // live user input and later strip or skip it when reasoning about the current turn.
42 const (
43 summaryTagOpen = "<compaction-summary>"
44 summaryTagClose = "</compaction-summary>"
45 )
46
47 // compactionInstruction is appended as the only new message after an otherwise
48 // byte-stable sampling prefix. This lets providers reuse the ordinary request's
49 // system, tools and message-prefix KV cache.
50 const compactionInstruction = `Compact the preceding conversation prefix into a durable resume briefing.
51 Write under these exact headings, omitting a heading only if it has no content:
52
53 ## Standing facts & constraints
54 Everything the user stated that still governs the work — names, paths, IDs, versions, tokens, preferences, and hard "never do X" rules — in their own words. Be exhaustive; this is the durable contract, so prefer over- to under-including.
55
56 ## Goal
57 The user's request and intent.
58
59 ## Decisions & rationale
60 Key choices made so far and why — so they are not re-litigated or reversed.
61
62 ## Files & code
63 Files read or modified, with the specific facts that matter: signatures, line locations, data shapes, and exact edits applied. Be concrete; this is what lets the agent act without re-reading everything.
64
65 ## Commands & outcomes
66 Commands run (builds, tests, git) and their relevant results — what passed, what failed, and the error text that matters.
67
68 ## Errors & fixes
69 Problems hit and how they were resolved (or not), so the same dead ends are not repeated.
70
71 ## Pending & next step
72 What is still in progress or unstarted, and the single most concrete next action to take.
73
74 Rules: be terse — bullet points and fragments, not prose. Preserve identifiers, paths, and numbers exactly. Merge valid facts from any existing <compaction-summary> and remove facts superseded by later messages. Do NOT invent anything not present in the messages; if something is unknown, leave it out rather than guessing. Output only the structured Markdown briefing. Do not call tools. Do not output reasoning.`
75
76 // compactTrigger is the sole automatic context-maintenance boundary. Output
77 // budgets are intentionally absent: they are clipped against the final request
78 // at send time and must never make compaction happen earlier than the user's
79 // configured compact_ratio.
80 func (a *Agent) compact(ctx context.Context, trigger, instructions string, force bool) error {
81 _, err := a.compactToProjectionWithChunked(ctx, trigger, instructions, foldRequest{
82 force: force, allowChunked: trigger == CompactionTriggerManual,
83 })
84 return err
85 }
86
87 func (a *Agent) compactToProjection(ctx context.Context, trigger, instructions string, force, mustFree bool) (CompactionOutcome, error) {
88 return a.compactToProjectionWithChunked(ctx, trigger, instructions, foldRequest{force: force, mustFree: mustFree})
89 }
90
91 func (a *Agent) compactToProjectionWithChunked(ctx context.Context, trigger, instructions string, req foldRequest) (outcome CompactionOutcome, err error) {
92 ctx, finish := a.beginCompactionRun(ctx)
93 defer func() { err = finish(err) }()
94 if err := a.sess.compactionRunMu.acquire(ctx); err != nil {
95 return CompactionNoop, err
96 }
97 defer a.sess.compactionRunMu.Unlock()
98 return a.compactToProjectionLocked(ctx, trigger, instructions, req)
99 }
100
101 func (a *Agent) compactTrigger() int {
102 window := a.effectiveContextWindow()
103 if a == nil || window <= 0 {
104 return 0
105 }
106 ratio := a.compactRatio
107 if ratio <= 0 {
108 ratio = defaultCompactRatio
109 }
110 if a.ablation.Off(ablation.Compaction) {
111 ratio = 0.5
112 }
113 return max(1, int(float64(window)*ratio))
114 }
115
116 // hardInputCeiling is a physical input-safety boundary, not another user
117 // compaction threshold. Reply budgets are resolved independently at send time.
118 func (a *Agent) hardInputCeiling() int {
119 window := a.effectiveContextWindow()
120 if a == nil || window <= 0 {
121 return 0
122 }
123 return max(1, window-protocolReserveTokens)
124 }
125
126 // recentTailBudget is the content-construction budget for the recent verbatim
127 // tail. Harness-style compaction always retains 16% of the model window.
128 func (a *Agent) recentTailBudget() int {
129 window := a.effectiveContextWindow()
130 if a == nil || window <= 0 {
131 return 1
132 }
133 return max(1, int(float64(window)*recentTailBudgetRatio))
134 }
135
136 // foldEconomics estimates whether compacting the given region saves enough
137 // tokens to justify the summarization API call. It returns false when the
138 // region is too small for the savings to outweigh the extra round-trip cost
139 // and latency of calling the summarizer.
140 func foldEconomics(region []provider.Message) bool {
141 const minFoldTokens = 400
142 return estimateMessagesTokens(region) >= minFoldTokens
143 }
144
145 func estimateMessagesTokens(msgs []provider.Message) int {
146 total := 0
147 for _, m := range msgs {
148 if m.LocalOnly || IsPinnedContextRevision(m) {
149 continue
150 }
151 total += 4 // chat-message framing overhead
152 total += estimateTextTokens(m.Content)
153 total += estimateTextTokens(m.ReasoningContent)
154 total += estimateTextTokens(m.Name)
155 total += estimateTextTokens(m.ToolCallID)
156 for _, tc := range m.ToolCalls {
157 total += 8
158 total += estimateTextTokens(tc.ID)
159 total += estimateTextTokens(tc.Name)
160 total += estimateTextTokens(tc.Arguments)
161 }
162 for _, item := range m.ResponsesItems {
163 total += estimateTextTokens(string(item))
164 }
165 for _, search := range m.ServerSearch {
166 provider.WalkServerSearchEstimate(search, func(s string) {
167 total += estimateTextTokens(s)
168 })
169 }
170 }
171 return total
172 }
173
174 func estimateTextTokens(s string) int {
175 if s == "" {
176 return 0
177 }
178 // A conservative cross-language approximation: English-ish text trends near
179 // four bytes per token, while CJK-heavy text is closer to one rune per token.
180 bytes := len(s)
181 runes := utf8.RuneCountInString(s)
182 byBytes := (bytes + 3) / 4
183 if runes > byBytes {
184 return runes
185 }
186 return byBytes
187 }
188
189 // SummarizeFrom keeps the compatibility index contract while installing a
190 // projection that compresses from that user-turn boundary onward.
191 func (a *Agent) SummarizeFrom(ctx context.Context, fromIdx int) error {
192 return a.summarizeAtProjectionBoundary(ctx, fromIdx, "after")
193 }
194
195 // SummarizeUpTo keeps the compatibility index contract while installing a
196 // projection that compresses everything before that user-turn boundary.
197 func (a *Agent) SummarizeUpTo(ctx context.Context, toIdx int) error {
198 return a.summarizeAtProjectionBoundary(ctx, toIdx, "before")
199 }
200
201 func (a *Agent) summarizeAtProjectionBoundary(ctx context.Context, canonicalIndex int, direction string) (resultErr error) {
202 ctx, finish := a.beginCompactionRun(ctx)
203 defer func() { resultErr = finish(resultErr) }()
204 if err := ctx.Err(); err != nil {
205 return err
206 }
207 snap := a.snapshotExplicitCompression()
208 if canonicalIndex < 0 || canonicalIndex >= len(snap.canonical) {
209 return nil
210 }
211 anchor := snap.canonical[canonicalIndex]
212 if !compressAnchorCandidate(anchor) {
213 return nil
214 }
215 visibleIndex := -1
216 for i, msg := range snap.visible {
217 if !compressAnchorCandidate(msg) {
218 continue
219 }
220 if anchor.CreatedAt != 0 && msg.CreatedAt == anchor.CreatedAt {
221 visibleIndex = i
222 break
223 }
224 if anchor.CreatedAt == 0 && UserMessageText(msg) == UserMessageText(anchor) {
225 if visibleIndex >= 0 {
226 return fmt.Errorf("summarize boundary is ambiguous in the current model context")
227 }
228 visibleIndex = i
229 }
230 }
231 if visibleIndex < 0 {
232 return fmt.Errorf("context compression unavailable: selected turn is no longer present in the model context")
233 }
234 result, err := a.compressVisibleRange(ctx, snap, CompactionTriggerManual, direction, visibleIndex, anchorPreview(UserMessageText(anchor)), "")
235 if err != nil {
236 return err
237 }
238 if result.Status != "ok" {
239 if result.Status == "noop" && noCompressionHistory(result.Reason) && result.SourceTokens < a.hardInputCeiling() {
240 return ctx.Err()
241 }
242 reason := strings.TrimSpace(result.Reason)
243 if reason == "" {
244 reason = "selected range did not reduce the model context"
245 }
246 return fmt.Errorf("context compression skipped: %s", reason)
247 }
248 return nil
249 }
250
251 // IsCompactionSummary reports whether m is a rolling digest inserted by a
252 // prior compaction fold. Exported for session owners outside this package
253 // (e.g. the guardian) whose turn rollback must not treat a digest as a
254 // disposable user message.
255 func IsCompactionSummary(m provider.Message) bool { return isCompactionSummary(m) }
256
257 func (a *Agent) activeTurnStart(msgs []provider.Message) int {
258 createdAt := a.activeTurnCreatedAt.Load()
259 if createdAt == 0 {
260 return -1
261 }
262 for i, m := range msgs {
263 if m.Role == provider.RoleUser && m.CreatedAt == createdAt {
264 return i
265 }
266 }
267 return -1
268 }
269
270 // isCompactionSummary reports whether m is a rolling summary from a prior fold.
271 func isCompactionSummary(m provider.Message) bool {
272 return m.Role == provider.RoleUser &&
273 strings.HasPrefix(strings.TrimLeft(m.Content, "\n "), summaryTagOpen)
274 }
275
276 // pinnedPrefixLen keeps only the system message. All older user turns,
277 // failures, and [[keep]] markers enter the Harness-style summary prefix.
278 func (a *Agent) pinnedPrefixLen(msgs []provider.Message) int {
279 if len(msgs) > 0 && msgs[0].Role == provider.RoleSystem {
280 return 1
281 }
282 return 0
283 }
284
285 // planCompaction returns [head:start] to fold while retaining the newest 16%
286 // of the model window and keeping tool-call/result groups balanced.
287 func (a *Agent) planCompaction(msgs []provider.Message, min int, force bool) (head, start int, ok bool) {
288 head = a.pinnedPrefixLen(msgs)
289 if a.contextWindow > 0 {
290 budget := a.recentTailBudget()
291 if force {
292 // Fixed instructions are not compressible history. Including them
293 // here can reserve the entire conversation as the recent tail and
294 // leave only a non-summarizable context snapshot in the fold.
295 _, history, _ := a.partitionFoldForProjectionAt(msgs[head:], head, latestSessionContextIndex(msgs))
296 if half := estimateMessagesTokens(modelInputMessages(history)) / 2; half > 0 && half < budget {
297 budget = half
298 }
299 }
300 start = tailStart(msgs, head, budget, a.tokPerChar(), a.tailFloor())
301 // Remeasure when force or non-strict roles; strict-alternating otherwise
302 // keeps a cheap tokPerChar overestimate of the tail under force.
303 floor := max(head, len(msgs)-a.tailFloor())
304 remeasure := force || !a.strictAlternatingRoles
305 for remeasure && start < floor && estimateMessagesTokens(provider.ModelMessages(msgs[start:])) > budget {
306 start++
307 for start < floor && start < len(msgs) && msgs[start].Role == provider.RoleTool {
308 start++
309 }
310 }
311 } else {
312 // No window: keep a fixed recent count, aligned off tool results.
313 start = len(msgs) - a.tailFloor()
314 for start > head && start < len(msgs) && msgs[start].Role == provider.RoleTool {
315 start--
316 }
317 }
318 start = max(start, head)
319 if start-head < min {
320 return head, start, false
321 }
322 return head, start, true
323 }
324
325 func (a *Agent) tailFloor() int {
326 return 0
327 }
328
329 // tailStart walks newest→oldest, growing the verbatim tail until the next
330 // message would push its token estimate past budgetTokens (but never below
331 // minKeep messages), then aligns the boundary back off any tool result so the
332 // tail never begins with an orphan whose assistant tool_calls were summarized
333 // away.
334 func tailStart(msgs []provider.Message, head, budgetTokens int, tokPerChar float64, minKeep int) int {
335 start := len(msgs)
336 acc := 0
337 for i := len(msgs) - 1; i > head; i-- {
338 c := int(float64(msgChars(msgs[i])) * tokPerChar)
339 if len(msgs)-i > minKeep && acc+c > budgetTokens {
340 break
341 }
342 acc += c
343 start = i
344 }
345 // start == len(msgs) when nothing fit the tail (a session too small to have a
346 // message after head); there is no msgs[start] to align off, and the caller's
347 // minCompactMessages check then no-ops the pass.
348 for start > head && start < len(msgs) && msgs[start].Role == provider.RoleTool {
349 start--
350 }
351 return start
352 }
353
354 // tokPerChar derives a tokens-per-character ratio from the last turn's real
355 // usage so per-message estimates track the provider's tokenizer without a local
356 // one. Reasoning content is excluded from the char count to match the prompt
357 // actually sent (the provider strips it). Falls back to ~4 chars/token before
358 // any usage is known, and ignores absurd ratios.
359 func (a *Agent) tokPerChar() float64 {
360 if cal := a.sess.output.promptCalibration.Load(); cal != nil && cal.compactChars > 0 {
361 if r := float64(cal.promptTokens) / float64(cal.compactChars); r > 0.05 && r < 2 {
362 return r
363 }
364 }
365 return fallbackTokPerChar
366 }
367
368 // msgChars counts the characters that ride to the provider for one message —
369 // content plus tool-call names and arguments, but not reasoning (stripped on
370 // send).
371 func msgChars(m provider.Message) int {
372 if m.LocalOnly {
373 return 0
374 }
375 n := len(m.Content)
376 for _, tc := range m.ToolCalls {
377 n += len(tc.Name) + len(tc.Arguments)
378 }
379 return n
380 }
381
382 func charsOfMessages(msgs []provider.Message) int {
383 n := 0
384 for _, m := range msgs {
385 n += msgChars(m)
386 }
387 return n
388 }
389
390 // summarize asks the executor's own provider to distill a replayed prefix into
391 // a briefing. instructions is optional /compact focus + PreCompact text.
392 // Named returns so defer can attach RequestCount and still return usage.
393 func compactionInstructionWithFocus(instructions string) string {
394 instruction := compactionInstruction
395 if strings.TrimSpace(instructions) != "" {
396 instruction += "\n\nAdditional focus for this compaction (prioritize keeping this):\n" + strings.TrimSpace(instructions)
397 }
398 return instruction
399 }
400
401 // summaryRequest builds the exact cache-aligned request shape used by
402 // summarize. Keeping planning and execution on this shared builder prevents a
403 // supposedly safe overflow fold from being rejected only after it is selected.
404 func (a *Agent) summaryRequest(region []provider.Message, instructions string) provider.Request {
405 prefix := append([]provider.Message(nil), region...)
406 for i := range prefix {
407 if !a.imageInput.native && prefix[i].VisionSummary != nil {
408 prefix[i].ImageInputs = nil
409 }
410 }
411 if len(prefix) == 0 || prefix[0].Role != provider.RoleSystem {
412 visible := a.modelVisibleMessages()
413 if len(visible) > 0 && visible[0].Role == provider.RoleSystem {
414 prefix = append([]provider.Message{visible[0]}, prefix...)
415 }
416 }
417 messages := a.normalizeModelRequestMessages(prefix)
418 messages = append(messages, HostGeneratedUserMessage(compactionInstructionWithFocus(instructions)))
419 var schemas []provider.ToolSchema
420 if a.svc.tools != nil {
421 schemas = a.providerToolSchemas()
422 }
423 return provider.Request{
424 Messages: messages,
425 Tools: schemas,
426 MaxTokens: a.summaryOutputBudget(),
427 Temperature: provider.OptionalTemperature(a.temperature),
428 }
429 }
430
431 // summarize asks the executor's own provider to distill a replayed prefix into
432 // a briefing. instructions is optional /compact focus + PreCompact text.
433 func (a *Agent) summarize(ctx context.Context, region []provider.Message, instructions string) (string, *provider.Usage, error) {
434 req := a.summaryRequest(region, instructions)
435 summary, usage, err := a.runSummaryRequest(ctx, req)
436 a.observeSummaryOutcome(req, usage, err)
437 return summary, usage, err
438 }
439
440 // runSummaryRequest admits, sends, and drains one summary request.
441 // Named returns so defer can attach RequestCount and still return usage.
442 func (a *Agent) runSummaryRequest(ctx context.Context, req provider.Request) (summary string, usage *provider.Usage, err error) {
443 if err := ctx.Err(); err != nil {
444 return "", nil, err
445 }
446 defer func(operationCtx context.Context) { err = summaryRequestError(operationCtx, err) }(ctx)
447 req.Messages, err = a.resolveRequestImages(ctx, req.Messages)
448 if err != nil {
449 return "", nil, err
450 }
451 ctx, cancel := context.WithCancel(ctx)
452 defer cancel()
453 ctx = provider.WithRequestAttemptCounter(ctx)
454 defer func() {
455 usage = provider.UsageWithRequestAttemptCount(ctx, usage)
456 if usage != nil && (usage.TotalTokens > 0 || usage.RequestCount > 0) {
457 a.svc.sink.Emit(event.Event{Kind: event.Usage, ModelRef: a.modelRef, Usage: usage, Pricing: a.svc.pricing, UsageSource: event.UsageSourceCompaction})
458 }
459 }()
460 defer trackPublishedHostStream(ctx, cancel)()
461 if err := a.applySummaryAdmissionToRequest(&req); err != nil {
462 return "", usage, err
463 }
464 if budget := a.summaryOutputBudget(); req.MaxTokens > budget {
465 req.MaxTokens = budget
466 }
467 if req.MaxTokens < 256 {
468 return "", usage, fmt.Errorf("summary output budget too small (%d tokens)", req.MaxTokens)
469 }
470 if a.svc.prov == nil {
471 return "", usage, fmt.Errorf("summary unavailable")
472 }
473 if err := ctx.Err(); err != nil {
474 return "", usage, err
475 }
476 ch, err := provider.StreamAuxiliary(ctx, a.svc.prov, req)
477 if err != nil {
478 return "", usage, err
479 }
480 defer func() {
481 cancel()
482 for range ch {
483 }
484 }()
485
486 // Cancel on timeout; join the buffer worker before releasing execution ownership.
487 var b strings.Builder
488 var reasoning strings.Builder
489 toolCalls := 0
490 for {
491 select {
492 case <-ctx.Done():
493 return "", usage, ctx.Err()
494 case chunk, ok := <-ch:
495 if !ok {
496 if usage != nil && usage.FinishReason == "length" {
497 return "", usage, fmt.Errorf("%w: provider reached the output token limit", errSummaryOutputTruncated)
498 }
499 s := strings.TrimSpace(b.String())
500 if s == "" {
501 // Thinking providers may answer with reasoning_content only. Surface
502 // it as the briefing unless the turn also reached for tools: that
503 // reasoning is private chain-of-thought, not digest material.
504 r := strings.TrimSpace(reasoning.String())
505 if r == "" || toolCalls > 0 {
506 return "", usage, errSummaryEmpty
507 }
508 return truncateUTF8Bytes(r, summaryReasoningMaxBytes), usage, nil
509 }
510 return s, usage, nil
511 }
512 switch chunk.Type {
513 case provider.ChunkText:
514 b.WriteString(chunk.Text)
515 case provider.ChunkReasoning:
516 reasoning.WriteString(chunk.Text)
517 case provider.ChunkToolCall, provider.ChunkToolCallStart:
518 toolCalls++
519 case provider.ChunkUsage:
520 usage = chunk.Usage
521 case provider.ChunkError:
522 return "", usage, chunk.Err
523 }
524 }
525 }
526 }
527
528 // summarizeOnce performs exactly one application-layer summary request.
529 // Timeouts, empty results, stream errors, and output truncation all fail once
530 // with no second attempt.
531 func (a *Agent) summarizeOnce(ctx context.Context, fold []provider.Message, instructions string) (string, *provider.Usage, error) {
532 return a.summarize(ctx, fold, instructions)
533 }
534
535 // renderTranscript flattens messages into a bounded transcript for the
536 // transcript-form summary request. Tool bodies are the provider-visible
537 // Content cut to slimToolResultRunes; RawContent never enters a summary.
538 func renderTranscript(msgs []provider.Message) string {
539 var b strings.Builder
540 for _, m := range msgs {
541 if m.LocalOnly {
542 continue
543 }
544 switch m.Role {
545 case provider.RoleUser:
546 fmt.Fprintf(&b, "[user]\n%s\n\n", m.Content)
547 case provider.RoleAssistant:
548 if m.Content != "" {
549 fmt.Fprintf(&b, "[assistant]\n%s\n", m.Content)
550 }
551 for _, tc := range m.ToolCalls {
552 fmt.Fprintf(&b, "[assistant calls %s] %s\n", tc.Name, summarizeToolArgs(tc.Arguments))
553 }
554 b.WriteString("\n")
555 case provider.RoleTool:
556 fmt.Fprintf(&b, "[tool %s result]\n%s\n\n", m.Name, slimToolResult(m.Content))
557 case provider.RoleSystem:
558 fmt.Fprintf(&b, "[system]\n%s\n\n", m.Content)
559 }
560 }
561 return b.String()
562 }
563
564 // summarizeToolArgs returns a short summary of tool-call arguments instead of
565 // the full JSON. This prevents the summarizer from reproducing long argument
566 // text (like sub-agent task prompts) in the compaction summary, which would
567 // leak into the session as a user message (#4317).
568 func summarizeToolArgs(args string) string {
569 if args == "" {
570 return "(no arguments)"
571 }
572 var parsed map[string]any
573 if err := json.Unmarshal([]byte(args), &parsed); err != nil {
574 // Not valid JSON — return a length hint instead of raw text.
575 return fmt.Sprintf("(%d bytes)", len(args))
576 }
577 keys := make([]string, 0, len(parsed))
578 for k := range parsed {
579 keys = append(keys, k)
580 }
581 sort.Strings(keys)
582 return fmt.Sprintf("{%s} (%d keys)", strings.Join(keys, ", "), len(parsed))
583 }
584
584 lines GO