返回 DeepSeek-Reasonix
prompt_truncation.go
根目录 / internal / agent / prompt_truncation.go
1 package agent
2
3 import (
4 "fmt"
5
6 "reasonix/internal/event"
7 "reasonix/internal/i18n"
8 )
9
10 const (
11 // truncationMinChars keeps short requests out of the detector: their token
12 // counts round too coarsely to demonstrate a ceiling.
13 truncationMinChars = 8192
14 // truncationMinCeiling is the smallest prompt size that can be a real
15 // context window. Below it the provider is reporting meaningless usage, not
16 // serving a small window, and nothing should be inferred from the number.
17 truncationMinCeiling = 512
18 // truncationDensityFloor is the tokens-per-character density below which no
19 // tokenizer operates. calibratedPromptTokens already refuses to learn from
20 // an observation this dense; read as evidence, it names why.
21 truncationDensityFloor = 0.05
22 // truncationCalibratedDrop is the share of an established density below
23 // which an observation cannot be the same tokenizer on the same wire.
24 truncationCalibratedDrop = 0.5
25 )
26
27 // promptTruncation is one provider that accepted a request and silently dropped
28 // part of the prompt: the ceiling it appears to enforce, and whether the session
29 // has been told. Grouped behind a single atomic pointer so the guarded state
30 // gains no scalar.
31 type promptTruncation struct {
32 promptCeiling int
33 notified bool
34 }
35
36 // promptTruncationCeiling reports the prompt ceiling a provider appears to have
37 // applied, or 0 when the reported size is consistent with the request that was
38 // sent.
39 //
40 // A truncating server answers HTTP 200 with no error field, so the only evidence
41 // is arithmetic: it reports far fewer prompt tokens than the characters on the
42 // wire can encode. Cache fields never deflate the count — CacheHitTokens is a
43 // subset of PromptTokens, not a deduction from it.
44 func promptTruncationCeiling(promptTokens int, shape requestCalibrationShape, cal *promptTokenCalibration) int {
45 if promptTokens < truncationMinCeiling || shape.requestChars < truncationMinChars {
46 return 0
47 }
48 density := float64(promptTokens) / float64(shape.requestChars)
49 // A calibration this model established is the sharper comparison, but only
50 // while it is itself plausible; otherwise fall through to the absolute floor
51 // rather than letting a bad calibration suppress detection entirely.
52 if cal != nil && cal.requestChars > 0 && cal.promptTokens > 0 {
53 if known := float64(cal.promptTokens) / float64(cal.requestChars); known > truncationDensityFloor {
54 if density < known*truncationCalibratedDrop {
55 return promptTokens
56 }
57 return 0
58 }
59 }
60 if density < truncationDensityFloor {
61 return promptTokens
62 }
63 return 0
64 }
65
66 // notePromptTruncation warns once per session, on the turn the truncation
67 // happens: the turn whose answer the user is about to read is the one the
68 // explanation belongs to.
69 //
70 // It deliberately does not clamp the window to the ceiling. Feeding it to
71 // learnContextBudget makes every later prompt that cannot fit fail admission
72 // outright, turning a session that degrades into one that stops — a much larger
73 // behavior change than detection, and one that needs its own evidence. Warning
74 // without clamping also keeps the cost of a wrong reading at one line of text.
75 func (a *Agent) notePromptTruncation(promptCeiling int) {
76 if a == nil || promptCeiling < truncationMinCeiling {
77 return
78 }
79 // Claim the warning with CAS: concurrent turns share this state, and the
80 // promise is one warning per session, not one per racing turn.
81 prev := a.sess.output.truncation.Load()
82 if prev != nil && prev.notified {
83 return
84 }
85 claimed := &promptTruncation{promptCeiling: promptCeiling, notified: true}
86 if !a.sess.output.truncation.CompareAndSwap(prev, claimed) {
87 return
88 }
89 if a.svc.sink == nil {
90 return
91 }
92 a.svc.sink.Emit(event.Event{
93 Kind: event.Notice,
94 Code: event.NoticeCodePromptTruncatedByServer,
95 Level: event.LevelWarn,
96 Text: i18n.M.PromptTruncatedByServerNotice,
97 Detail: fmt.Sprintf("accepted_prompt_tokens=%d", promptCeiling),
98 })
99 }
100
100 lines GO