返回 DeepSeek-Reasonix
output_budget.go
根目录 / internal / agent / output_budget.go
1 package agent
2
3 import (
4 "fmt"
5 "math"
6 "strings"
7 "sync"
8 "sync/atomic"
9 "time"
10 "unicode/utf8"
11
12 "reasonix/internal/nilutil"
13 "reasonix/internal/provider"
14 )
15
16 const (
17 outputBudgetReserve = 8 * 1024
18 minOutputBudgetReserve = protocolReserveTokens
19 )
20
21 const learnedOutputBudgetTTL = 24 * time.Hour
22
23 type learnedOutputBudgetCacheEntry struct {
24 completionBudget int
25 expiresAt time.Time
26 }
27
28 var learnedOutputBudgetCache = struct {
29 sync.Mutex
30 entries map[string]learnedOutputBudgetCacheEntry
31 }{entries: make(map[string]learnedOutputBudgetCacheEntry)}
32
33 type outputBudgetState struct {
34 outputBudget int
35 // lastUsage caches the latest provider telemetry for per-turn readouts.
36 // The run loop writes it while a frontend reads it, so it is atomic.
37 lastUsage atomic.Pointer[provider.Usage]
38 activeReqShape atomic.Pointer[requestCalibrationShape]
39 promptCalibration atomic.Pointer[promptTokenCalibration]
40 contextUsage atomic.Pointer[contextUsage] // gauge's memoised prompt size
41 learned atomic.Pointer[learnedContextBudget]
42 admission atomic.Pointer[contextAdmission]
43 truncation atomic.Pointer[promptTruncation]
44 }
45
46 // learnedContextBudget is an Agent-local observation of the live provider/model
47 // window. Completion limits are additionally shared through the short-lived
48 // provider/model cache below so a model rebuild does not immediately repeat a
49 // known over-limit request. The cache is deliberately in-memory and expires
50 // after one day; it is not a persisted global provider limit.
51 type learnedContextBudget struct {
52 windowTokens int
53 completionBudget int
54 }
55
56 const (
57 contextRecoveryNone = "none"
58 contextRecoveryProactiveClip = "proactive_clip"
59 contextRecoveryLearnedRetry = "learned_retry"
60 contextRecoveryCompacted = "compacted"
61 contextRecoveryFailed = "failed"
62 )
63
64 type contextAdmission struct {
65 WindowMode string
66 LimitMode string
67 Source string
68 WindowTokens int
69 PromptTokens int
70 AutoOutputTokens int
71 MaxOutputTokens int
72 RequestedOutputTokens int
73 EffectiveOutputTokens int
74 ReserveTokens int
75 PhysicalRemaining int
76 Clipped bool
77 ApplyMaxTokens bool
78 LastRecovery string
79 ObservedWindow int
80 ObservedPrompt int
81 ObservedCompletion int
82 }
83
84 type promptTokenCalibration struct {
85 promptTokens int
86 requestChars int64
87 compactChars int64
88 cjkRunes int64
89 cjkBytes int64
90 }
91
92 // requestCalibrationShape pairs the conservative provider-visible text and CJK
93 // composition used for overflow protection with the legacy content-only shape
94 // used by fold economics. Keeping them in one immutable pointer ensures readers
95 // never combine calibration fields from different prepared requests.
96 type requestCalibrationShape struct {
97 requestChars int64
98 compactChars int64
99 cjkRunes int64
100 cjkBytes int64
101 }
102
103 // reset drops what belongs to the transcript being replaced. Prompt-token
104 // calibration and the learned window are properties of the bound model/provider,
105 // and a model switch rebuilds the Agent, so they outlive SetSession. Admission
106 // describes one transcript's latest request and must never bleed into the next.
107 func (o *outputBudgetState) reset() {
108 o.lastUsage.Store(nil)
109 o.activeReqShape.Store(nil)
110 o.admission.Store(nil)
111 o.contextUsage.Store(nil)
112 }
113
114 func (a *Agent) setPromptTokenCalibration(promptTokens int, shape requestCalibrationShape) {
115 if a == nil || promptTokens <= 0 || shape.requestChars <= 0 {
116 return
117 }
118 // A prompt the provider truncated describes its own ceiling, not this
119 // model's tokenizer. Learning from it shrinks every later estimate, which
120 // delays compaction and truncates more of the next prompt.
121 if ceiling := promptTruncationCeiling(promptTokens, shape, a.sess.output.promptCalibration.Load()); ceiling > 0 {
122 a.notePromptTruncation(ceiling)
123 return
124 }
125 a.sess.output.promptCalibration.Store(&promptTokenCalibration{
126 promptTokens: promptTokens,
127 requestChars: shape.requestChars,
128 compactChars: shape.compactChars,
129 cjkRunes: shape.cjkRunes,
130 cjkBytes: shape.cjkBytes,
131 })
132 }
133
134 func (a *Agent) setPromptTokenCalibrationFromActive(promptTokens int) {
135 if a == nil {
136 return
137 }
138 if shape := a.sess.output.activeReqShape.Load(); shape != nil {
139 a.setPromptTokenCalibration(promptTokens, *shape)
140 }
141 }
142
143 // setPromptTokenCalibrationFromUsage trusts provider telemetry only;
144 // reconstructed usage remains available for accounting but not admission.
145 func (a *Agent) setPromptTokenCalibrationFromUsage(usage *provider.Usage) {
146 if a == nil || usage == nil || usage.Estimated {
147 return
148 }
149 a.setPromptTokenCalibrationFromActive(usage.LatestPromptTokens())
150 }
151
152 func outputBudgetOf(p provider.Provider) int {
153 if nilutil.IsNil(p) {
154 return 0
155 }
156 if budget, ok := p.(provider.OutputBudgetProvider); ok {
157 return budget.OutputBudget()
158 }
159 return 0
160 }
161
162 func sharesContextWindow(p provider.Provider) bool {
163 return contextBudgetPolicyOf(p).WindowMode == provider.ContextWindowShared
164 }
165
166 func contextBudgetPolicyOf(p provider.Provider) provider.ContextBudgetPolicy {
167 if nilutil.IsNil(p) {
168 return provider.ContextBudgetPolicy{}
169 }
170 return provider.ResolveContextBudgetPolicy(p)
171 }
172
173 func (a *Agent) learnOutputBudget(limit int) {
174 if a == nil || limit <= 0 {
175 return
176 }
177 cacheLearnedOutputBudget(outputBudgetCacheKey(a), limit)
178 current := a.sess.output.learned.Load()
179 if current != nil && current.completionBudget > 0 && current.completionBudget <= limit {
180 return
181 }
182 learned := &learnedContextBudget{completionBudget: limit}
183 if current != nil {
184 learned.windowTokens = current.windowTokens
185 }
186 a.sess.output.learned.Store(learned)
187 }
188
189 func outputBudgetCacheKey(a *Agent) string {
190 if a == nil {
191 return ""
192 }
193 providerName := ""
194 if !nilutil.IsNil(a.svc.prov) {
195 providerName = strings.TrimSpace(a.svc.prov.Name())
196 }
197 modelRef := strings.TrimSpace(a.modelRef)
198 if providerName == "" && modelRef == "" {
199 return ""
200 }
201 // Provider names are route-specific for the built-in OpenCode Go entries;
202 // retaining modelRef as a second component keeps custom routes isolated too.
203 return providerName + "|" + modelRef
204 }
205
206 func cacheLearnedOutputBudget(key string, limit int) {
207 if strings.TrimSpace(key) == "" || limit <= 0 {
208 return
209 }
210 now := time.Now()
211 learnedOutputBudgetCache.Lock()
212 defer learnedOutputBudgetCache.Unlock()
213 if current, ok := learnedOutputBudgetCache.entries[key]; ok && current.expiresAt.After(now) && current.completionBudget > 0 && current.completionBudget <= limit {
214 return
215 }
216 learnedOutputBudgetCache.entries[key] = learnedOutputBudgetCacheEntry{
217 completionBudget: limit,
218 expiresAt: now.Add(learnedOutputBudgetTTL),
219 }
220 }
221
222 func cachedLearnedOutputBudget(key string) int {
223 if strings.TrimSpace(key) == "" {
224 return 0
225 }
226 now := time.Now()
227 learnedOutputBudgetCache.Lock()
228 defer learnedOutputBudgetCache.Unlock()
229 entry, ok := learnedOutputBudgetCache.entries[key]
230 if !ok {
231 return 0
232 }
233 if !entry.expiresAt.After(now) {
234 delete(learnedOutputBudgetCache.entries, key)
235 return 0
236 }
237 return entry.completionBudget
238 }
239
240 func sharedWindowInputPolicyOf(p provider.Provider) provider.SharedWindowInputPolicy {
241 if nilutil.IsNil(p) {
242 return provider.SharedWindowInputPolicy{}
243 }
244 policy, ok := p.(provider.SharedWindowInputPolicyProvider)
245 if !ok {
246 return provider.SharedWindowInputPolicy{}
247 }
248 return policy.SharedWindowInputPolicy()
249 }
250
251 func requestCalibrationShapeOf(req provider.Request) requestCalibrationShape {
252 return requestCalibrationShapeWithPolicy(req, provider.SharedWindowInputPolicy{})
253 }
254
255 func (a *Agent) requestCalibrationShape(req provider.Request) requestCalibrationShape {
256 return requestCalibrationShapeWithPolicy(req, sharedWindowInputPolicyOf(a.svc.prov))
257 }
258
259 func requestCalibrationShapeWithPolicy(req provider.Request, policy provider.SharedWindowInputPolicy) requestCalibrationShape {
260 requestChars, cjkRunes, cjkBytes := requestCalibrationTextShape(req, policy)
261 return requestCalibrationShape{
262 requestChars: requestChars,
263 compactChars: int64(charsOfMessages(req.Messages)),
264 cjkRunes: cjkRunes,
265 cjkBytes: cjkBytes,
266 }
267 }
268
269 // requestCalibrationTextShape counts common shared-window text plus only the
270 // adapter-specific replay fields declared by the active provider. This keeps
271 // omitted bytes out of the ratio without missing newly appended wire content.
272 func requestCalibrationTextShape(req provider.Request, policy provider.SharedWindowInputPolicy) (chars, cjkRunes, cjkBytes int64) {
273 add := func(s string) {
274 chars += int64(len(s))
275 for _, r := range s {
276 if isCJKRune(r) {
277 cjkRunes++
278 cjkBytes += int64(utf8.RuneLen(r))
279 }
280 }
281 }
282 for _, msg := range req.Messages {
283 if msg.LocalOnly {
284 continue
285 }
286 chars += 4
287 add(string(msg.Role))
288 add(msg.Content)
289 if msg.Role == provider.RoleAssistant && (len(msg.ToolCalls) > 0 || policy.ReplaysOrdinaryReasoning) {
290 add(msg.ReasoningContent)
291 }
292 add(msg.Name)
293 add(msg.ToolCallID)
294 for _, call := range msg.ToolCalls {
295 chars += 8
296 add(call.ID)
297 add(call.Name)
298 add(call.Arguments)
299 }
300 if policy.ReplaysResponsesItems {
301 for _, item := range msg.ResponsesItems {
302 add(string(item))
303 }
304 }
305 for _, search := range msg.ServerSearch {
306 provider.WalkServerSearchEstimate(search, add)
307 }
308 }
309 for _, schema := range req.Tools {
310 chars += 8
311 add(schema.Name)
312 add(schema.Description)
313 add(string(schema.Parameters))
314 }
315 return chars, cjkRunes, cjkBytes
316 }
317
318 func (a *Agent) calibratedPromptTokens(shape requestCalibrationShape) (int, bool) {
319 if shape.requestChars <= 0 {
320 return 0, false
321 }
322 if cal := a.sess.output.promptCalibration.Load(); cal != nil && cal.requestChars > 0 {
323 ratio := float64(cal.promptTokens) / float64(cal.requestChars)
324 if ratio > 0.05 && ratio < 2 {
325 trustedChars := shape.requestChars
326 excessCJKBytes := int64(0)
327 // A higher CJK share cannot safely reuse the aggregate ratio. Scale its
328 // represented share and price only the excess at the cold rate,
329 // preserving exact calibration for stable CJK sessions.
330 if shape.cjkRunes*cal.requestChars > cal.cjkRunes*shape.requestChars {
331 trustedCJKBytes := min(cal.cjkBytes*shape.requestChars/cal.requestChars, shape.cjkBytes)
332 excessCJKBytes = shape.cjkBytes - trustedCJKBytes
333 trustedChars -= excessCJKBytes
334 }
335 cold := math.Ceil(float64(excessCJKBytes) * fallbackTokPerChar)
336 return int(math.Ceil(float64(trustedChars)*ratio) + cold), true
337 }
338 }
339 return 0, false
340 }
341
342 // estimatedPromptTokens sizes the provider-visible messages in real tokens —
343 // the only unit comparable against the context window. Same-session usage
344 // calibrates it; before that the wire character count carries the ~4 chars per
345 // token shape. estimateMessagesTokens counts characters and is for internal
346 // planning budgets only; against the window it would compact 4x early.
347 func (a *Agent) estimatedPromptTokens(msgs []provider.Message) int {
348 return a.estimatedShapeTokens(a.requestCalibrationShape(provider.Request{Messages: msgs}))
349 }
350
351 func (a *Agent) estimatedRequestTokens(req provider.Request) int {
352 return a.estimatedShapeTokens(a.requestCalibrationShape(req))
353 }
354
355 func (a *Agent) estimatedShapeTokens(shape requestCalibrationShape) int {
356 if shape.requestChars <= 0 {
357 return 0
358 }
359 if calibrated, ok := a.calibratedPromptTokens(shape); ok {
360 return calibrated
361 }
362 return int(float64(shape.requestChars) * fallbackTokPerChar)
363 }
364
365 func isCJKRune(r rune) bool {
366 return (r >= 0x4E00 && r <= 0x9FFF) ||
367 (r >= 0x3400 && r <= 0x4DBF) ||
368 (r >= 0x3040 && r <= 0x30FF) ||
369 (r >= 0xAC00 && r <= 0xD7AF)
370 }
371
372 func (a *Agent) effectiveContextWindow() int {
373 if a == nil {
374 return 0
375 }
376 cfg := a.contextWindow
377 learned := 0
378 if snap := a.sess.output.learned.Load(); snap != nil {
379 learned = snap.windowTokens
380 }
381 switch {
382 case cfg > 0 && learned > 0:
383 return min(cfg, learned)
384 case learned > 0:
385 return learned
386 default:
387 return cfg
388 }
389 }
390
391 func (a *Agent) learnedCompletionBudget() int {
392 if a == nil {
393 return 0
394 }
395 cached := cachedLearnedOutputBudget(outputBudgetCacheKey(a))
396 if snap := a.sess.output.learned.Load(); snap != nil {
397 if cached > 0 && (snap.completionBudget <= 0 || cached < snap.completionBudget) {
398 a.learnOutputBudget(cached)
399 return cached
400 }
401 return snap.completionBudget
402 }
403 if cached > 0 {
404 a.learnOutputBudget(cached)
405 return cached
406 }
407 return 0
408 }
409
410 func (a *Agent) learnContextBudget(window, completion int, omittedOutput bool) {
411 if a == nil {
412 return
413 }
414 cur := learnedContextBudget{}
415 if prev := a.sess.output.learned.Load(); prev != nil {
416 cur = *prev
417 }
418 if window > 0 {
419 if cur.windowTokens <= 0 || window < cur.windowTokens {
420 cur.windowTokens = window
421 }
422 }
423 if omittedOutput && completion > 0 {
424 cur.completionBudget = completion
425 }
426 next := cur
427 a.sess.output.learned.Store(&next)
428 }
429
430 func (a *Agent) storeAdmission(adm contextAdmission) {
431 if a == nil {
432 return
433 }
434 cp := adm
435 a.sess.output.admission.Store(&cp)
436 }
437
438 func (a *Agent) lastAdmission() contextAdmission {
439 if a == nil {
440 return contextAdmission{LastRecovery: contextRecoveryNone}
441 }
442 if snap := a.sess.output.admission.Load(); snap != nil {
443 return *snap
444 }
445 return contextAdmission{LastRecovery: contextRecoveryNone}
446 }
447
448 func (a *Agent) setLastRecovery(kind string) {
449 if a == nil {
450 return
451 }
452 adm := a.lastAdmission()
453 adm.LastRecovery = kind
454 a.storeAdmission(adm)
455 }
456
457 func admissionSource(userMax int, policy provider.ContextBudgetPolicy, learnedWindow bool) string {
458 if learnedWindow {
459 return provider.ContextBudgetSourceLearned
460 }
461 if userMax > 0 {
462 return provider.ContextBudgetSourceExplicit
463 }
464 switch {
465 case policy.AutoOutputTokens == provider.DeepSeekMaxOutputTokens && policy.LimitMode == provider.OutputLimitOmitWhenSafe:
466 return provider.ContextBudgetSourceOfficial
467 case policy.LimitMode == provider.OutputLimitAlways && policy.MaxOutputTokens > 0:
468 return provider.ContextBudgetSourceOpenCode
469 case policy.WindowMode == provider.ContextWindowUnknown || policy.AutoOutputTokens <= 0:
470 return provider.ContextBudgetSourceUnknown
471 default:
472 return provider.ContextBudgetSourceOfficial
473 }
474 }
475
476 // effectiveOutputBudget clips completion tokens at send time only; it never
477 // moves compact_ratio. Calibrated exhausted windows fail locally; a cold
478 // estimate that differs from the provider tokenizer uses bounded 400 recovery.
479 func (a *Agent) effectiveOutputBudget(req provider.Request) (int, bool, error) {
480 adm, err := a.admitOutputBudget(req)
481 if err != nil {
482 return 0, false, err
483 }
484 if !adm.ApplyMaxTokens || !adm.Clipped {
485 if adm.ApplyMaxTokens && adm.EffectiveOutputTokens > 0 && !adm.Clipped {
486 return adm.EffectiveOutputTokens, false, nil
487 }
488 return 0, false, nil
489 }
490 return adm.EffectiveOutputTokens, true, nil
491 }
492
493 func (a *Agent) admitOutputBudget(req provider.Request) (contextAdmission, error) {
494 return a.admitOutputBudgetWithReserve(req, outputBudgetReserveForWindow(a.effectiveContextWindow()), false)
495 }
496
497 // outputBudgetReserveForWindow preserves the original safety ratio: the 8K
498 // tokenizer/protocol cushion was chosen for a 1M-token window, so smaller
499 // windows reserve roughly the same 1/128 share. A 256-token floor covers
500 // framing, while the 8K cap keeps larger windows byte-stable.
501 func outputBudgetReserveForWindow(window int) int {
502 if window <= 0 {
503 return outputBudgetReserve
504 }
505 return min(outputBudgetReserve, max(minOutputBudgetReserve, window/128))
506 }
507
508 // admitSummaryOutputBudget uses the summary request's dedicated protocol
509 // reserve instead of the ordinary-turn reserve. Unknown gateways are treated
510 // as shared when an effective window exists: summary planning already makes
511 // that conservative assumption, so execution must enforce the same contract.
512 func (a *Agent) admitSummaryOutputBudget(req provider.Request) (contextAdmission, error) {
513 return a.admitOutputBudgetWithReserve(req, protocolReserveTokens, true)
514 }
515
516 func shouldUseSharedWindowForAdmission(mode provider.ContextWindowMode, observedWindow int, conservativeUnknown bool) bool {
517 return mode == provider.ContextWindowUnknown && (observedWindow > 0 || conservativeUnknown)
518 }
519
520 func (a *Agent) admitOutputBudgetWithReserve(req provider.Request, reserveTokens int, conservativeUnknown bool) (contextAdmission, error) {
521 adm := contextAdmission{
522 ReserveTokens: reserveTokens,
523 LastRecovery: a.lastAdmission().LastRecovery,
524 Source: provider.ContextBudgetSourceUnknown,
525 }
526 if adm.LastRecovery == "" {
527 adm.LastRecovery = contextRecoveryNone
528 }
529 if a == nil {
530 return adm, nil
531 }
532 if learned := a.sess.output.learned.Load(); learned != nil {
533 adm.ObservedWindow = learned.windowTokens
534 adm.ObservedCompletion = learned.completionBudget
535 }
536 policy := contextBudgetPolicyOf(a.svc.prov)
537 if shouldUseSharedWindowForAdmission(policy.WindowMode, adm.ObservedWindow, conservativeUnknown) {
538 policy.WindowMode = provider.ContextWindowShared
539 }
540 if policy.AutoOutputTokens <= 0 && a.learnedCompletionBudget() > 0 {
541 policy.AutoOutputTokens = a.learnedCompletionBudget()
542 }
543 if learned := a.learnedCompletionBudget(); learned > 0 {
544 if policy.AutoOutputTokens <= 0 || learned < policy.AutoOutputTokens {
545 policy.AutoOutputTokens = learned
546 }
547 if policy.MaxOutputTokens <= 0 || learned < policy.MaxOutputTokens {
548 policy.MaxOutputTokens = learned
549 }
550 }
551 adm.WindowMode = policy.WindowMode.String()
552 adm.LimitMode = policy.LimitMode.String()
553 adm.AutoOutputTokens = policy.AutoOutputTokens
554 adm.MaxOutputTokens = policy.MaxOutputTokens
555 window := a.effectiveContextWindow()
556 adm.WindowTokens = window
557 learnedWindow := window > 0 && (a.contextWindow <= 0 || window < a.contextWindow)
558 adm.Source = admissionSource(req.MaxTokens, policy, learnedWindow)
559 if window <= 0 {
560 a.storeAdmission(adm)
561 return adm, nil
562 }
563 est := a.estimatedRequestTokens(req)
564 adm.PromptTokens = est
565 physical := window - est - reserveTokens
566 adm.PhysicalRemaining = physical
567 shared := policy.WindowMode == provider.ContextWindowShared
568 if !shared {
569 a.applyLimitMode(&adm, req.MaxTokens, policy, physical)
570 a.storeAdmission(adm)
571 return adm, nil
572 }
573 if physical <= 0 {
574 a.storeAdmission(adm)
575 return adm, fmt.Errorf("%w: estimated prompt %d leaves no shared-window output budget", ErrCompactionRequired, est)
576 }
577 requested := 0
578 switch {
579 case req.MaxTokens > 0:
580 requested = req.MaxTokens
581 default:
582 requested = policy.AutoOutputTokens
583 }
584 if policy.MaxOutputTokens > 0 && requested > policy.MaxOutputTokens {
585 requested = policy.MaxOutputTokens
586 }
587 adm.RequestedOutputTokens = requested
588 if req.MaxTokens < 0 {
589 if requested > 0 && requested > physical {
590 a.storeAdmission(adm)
591 return adm, fmt.Errorf("%w: estimated prompt %d leaves no room for omitted auto output %d", ErrCompactionRequired, est, requested)
592 }
593 a.storeAdmission(adm)
594 return adm, nil
595 }
596 if requested <= 0 {
597 a.applyLimitMode(&adm, req.MaxTokens, policy, physical)
598 a.storeAdmission(adm)
599 return adm, nil
600 }
601 effective := requested
602 if effective > physical {
603 effective = physical
604 adm.Clipped = true
605 }
606 adm.EffectiveOutputTokens = effective
607 a.applyLimitMode(&adm, req.MaxTokens, policy, physical)
608 if adm.Clipped {
609 adm.ApplyMaxTokens = req.MaxTokens >= 0 && policy.LimitMode != provider.OutputLimitUnsupported
610 adm.EffectiveOutputTokens = effective
611 }
612 if adm.Clipped && adm.LastRecovery == contextRecoveryNone {
613 adm.LastRecovery = contextRecoveryProactiveClip
614 }
615 a.storeAdmission(adm)
616 return adm, nil
617 }
618
619 func (a *Agent) applyAdmissionToRequest(req *provider.Request) error {
620 if a == nil || req == nil {
621 return nil
622 }
623 adm, err := a.admitOutputBudget(*req)
624 if err != nil {
625 return err
626 }
627 if adm.ApplyMaxTokens && adm.EffectiveOutputTokens > 0 {
628 req.MaxTokens = adm.EffectiveOutputTokens
629 }
630 return nil
631 }
632
633 func (a *Agent) applySummaryAdmissionToRequest(req *provider.Request) error {
634 if a == nil || req == nil {
635 return nil
636 }
637 adm, err := a.admitSummaryOutputBudget(*req)
638 if err != nil {
639 return err
640 }
641 if adm.ApplyMaxTokens && adm.EffectiveOutputTokens > 0 {
642 req.MaxTokens = adm.EffectiveOutputTokens
643 }
644 return nil
645 }
646
647 func (a *Agent) applyLimitMode(adm *contextAdmission, userMax int, policy provider.ContextBudgetPolicy, physical int) {
648 if userMax < 0 || policy.LimitMode == provider.OutputLimitUnsupported {
649 adm.ApplyMaxTokens = false
650 return
651 }
652 effective := adm.EffectiveOutputTokens
653 if effective <= 0 {
654 if userMax > 0 {
655 effective = userMax
656 } else {
657 effective = policy.AutoOutputTokens
658 }
659 if policy.MaxOutputTokens > 0 && effective > policy.MaxOutputTokens {
660 effective = policy.MaxOutputTokens
661 }
662 if policy.WindowMode == provider.ContextWindowShared && physical > 0 && effective > physical {
663 effective = physical
664 adm.Clipped = true
665 }
666 }
667 switch policy.LimitMode {
668 case provider.OutputLimitAlways, provider.OutputLimitRequired:
669 if effective > 0 {
670 adm.ApplyMaxTokens = true
671 adm.EffectiveOutputTokens = effective
672 }
673 case provider.OutputLimitOmitWhenSafe:
674 if userMax > 0 || adm.Clipped {
675 adm.ApplyMaxTokens = true
676 adm.EffectiveOutputTokens = effective
677 }
678 }
679 }
680
680 lines GO