| 1 | package agent |
| 2 | |
| 3 | import ( |
| 4 | "fmt" |
| 5 | |
| 6 | "reasonix/internal/event" |
| 7 | "reasonix/internal/i18n" |
| 8 | ) |
| 9 | |
| 10 | const ( |
| 11 | // truncationMinChars keeps short requests out of the detector: their token |
| 12 | // counts round too coarsely to demonstrate a ceiling. |
| 13 | truncationMinChars = 8192 |
| 14 | // truncationMinCeiling is the smallest prompt size that can be a real |
| 15 | // context window. Below it the provider is reporting meaningless usage, not |
| 16 | // serving a small window, and nothing should be inferred from the number. |
| 17 | truncationMinCeiling = 512 |
| 18 | // truncationDensityFloor is the tokens-per-character density below which no |
| 19 | // tokenizer operates. calibratedPromptTokens already refuses to learn from |
| 20 | // an observation this dense; read as evidence, it names why. |
| 21 | truncationDensityFloor = 0.05 |
| 22 | // truncationCalibratedDrop is the share of an established density below |
| 23 | // which an observation cannot be the same tokenizer on the same wire. |
| 24 | truncationCalibratedDrop = 0.5 |
| 25 | ) |
| 26 | |
| 27 | // promptTruncation is one provider that accepted a request and silently dropped |
| 28 | // part of the prompt: the ceiling it appears to enforce, and whether the session |
| 29 | // has been told. Grouped behind a single atomic pointer so the guarded state |
| 30 | // gains no scalar. |
| 31 | type promptTruncation struct { |
| 32 | promptCeiling int |
| 33 | notified bool |
| 34 | } |
| 35 | |
| 36 | // promptTruncationCeiling reports the prompt ceiling a provider appears to have |
| 37 | // applied, or 0 when the reported size is consistent with the request that was |
| 38 | // sent. |
| 39 | // |
| 40 | // A truncating server answers HTTP 200 with no error field, so the only evidence |
| 41 | // is arithmetic: it reports far fewer prompt tokens than the characters on the |
| 42 | // wire can encode. Cache fields never deflate the count — CacheHitTokens is a |
| 43 | // subset of PromptTokens, not a deduction from it. |
| 44 | func promptTruncationCeiling(promptTokens int, shape requestCalibrationShape, cal *promptTokenCalibration) int { |
| 45 | if promptTokens < truncationMinCeiling || shape.requestChars < truncationMinChars { |
| 46 | return 0 |
| 47 | } |
| 48 | density := float64(promptTokens) / float64(shape.requestChars) |
| 49 | // A calibration this model established is the sharper comparison, but only |
| 50 | // while it is itself plausible; otherwise fall through to the absolute floor |
| 51 | // rather than letting a bad calibration suppress detection entirely. |
| 52 | if cal != nil && cal.requestChars > 0 && cal.promptTokens > 0 { |
| 53 | if known := float64(cal.promptTokens) / float64(cal.requestChars); known > truncationDensityFloor { |
| 54 | if density < known*truncationCalibratedDrop { |
| 55 | return promptTokens |
| 56 | } |
| 57 | return 0 |
| 58 | } |
| 59 | } |
| 60 | if density < truncationDensityFloor { |
| 61 | return promptTokens |
| 62 | } |
| 63 | return 0 |
| 64 | } |
| 65 | |
| 66 | // notePromptTruncation warns once per session, on the turn the truncation |
| 67 | // happens: the turn whose answer the user is about to read is the one the |
| 68 | // explanation belongs to. |
| 69 | // |
| 70 | // It deliberately does not clamp the window to the ceiling. Feeding it to |
| 71 | // learnContextBudget makes every later prompt that cannot fit fail admission |
| 72 | // outright, turning a session that degrades into one that stops — a much larger |
| 73 | // behavior change than detection, and one that needs its own evidence. Warning |
| 74 | // without clamping also keeps the cost of a wrong reading at one line of text. |
| 75 | func (a *Agent) notePromptTruncation(promptCeiling int) { |
| 76 | if a == nil || promptCeiling < truncationMinCeiling { |
| 77 | return |
| 78 | } |
| 79 | // Claim the warning with CAS: concurrent turns share this state, and the |
| 80 | // promise is one warning per session, not one per racing turn. |
| 81 | prev := a.sess.output.truncation.Load() |
| 82 | if prev != nil && prev.notified { |
| 83 | return |
| 84 | } |
| 85 | claimed := &promptTruncation{promptCeiling: promptCeiling, notified: true} |
| 86 | if !a.sess.output.truncation.CompareAndSwap(prev, claimed) { |
| 87 | return |
| 88 | } |
| 89 | if a.svc.sink == nil { |
| 90 | return |
| 91 | } |
| 92 | a.svc.sink.Emit(event.Event{ |
| 93 | Kind: event.Notice, |
| 94 | Code: event.NoticeCodePromptTruncatedByServer, |
| 95 | Level: event.LevelWarn, |
| 96 | Text: i18n.M.PromptTruncatedByServerNotice, |
| 97 | Detail: fmt.Sprintf("accepted_prompt_tokens=%d", promptCeiling), |
| 98 | }) |
| 99 | } |
| 100 |