| 1 | package agent |
| 2 | |
| 3 | import ( |
| 4 | "sync" |
| 5 | "testing" |
| 6 | ) |
| 7 | |
| 8 | // The measured Ollama case: a ~12,000-token prompt answered with HTTP 200 and |
| 9 | // prompt_tokens 2051. Nothing else in the response says the prompt was cut. |
| 10 | func TestPromptTruncationCeilingDetectsTheMeasuredOllamaCase(t *testing.T) { |
| 11 | shape := requestCalibrationShape{requestChars: 54_000} |
| 12 | if got := promptTruncationCeiling(2051, shape, nil); got != 2051 { |
| 13 | t.Fatalf("ceiling = %d, want 2051", got) |
| 14 | } |
| 15 | } |
| 16 | |
| 17 | // The case the absolute floor alone misses: a density of 0.10 is a plausible |
| 18 | // tokenizer in isolation, so it is accepted and becomes the calibration. Against |
| 19 | // an established 0.26 it cannot be the same tokenizer on the same wire. |
| 20 | func TestPromptTruncationCeilingCatchesADropFromAnEstablishedDensity(t *testing.T) { |
| 21 | cal := &promptTokenCalibration{promptTokens: 2600, requestChars: 10_000} |
| 22 | shape := requestCalibrationShape{requestChars: 20_000} |
| 23 | if got := promptTruncationCeiling(2051, shape, cal); got != 2051 { |
| 24 | t.Fatalf("ceiling = %d, want 2051", got) |
| 25 | } |
| 26 | } |
| 27 | |
| 28 | func TestPromptTruncationCeilingLeavesHonestObservationsAlone(t *testing.T) { |
| 29 | cal := &promptTokenCalibration{promptTokens: 2600, requestChars: 10_000} |
| 30 | for _, tc := range []struct { |
| 31 | name string |
| 32 | promptTokens int |
| 33 | chars int64 |
| 34 | cal *promptTokenCalibration |
| 35 | }{ |
| 36 | {"uncalibrated dense CJK", 4_000, 20_000, nil}, |
| 37 | {"calibrated steady", 5_200, 20_000, cal}, |
| 38 | {"calibrated mild drift", 3_200, 20_000, cal}, |
| 39 | {"request too short to judge", 200, 4_000, nil}, |
| 40 | {"no usage reported", 0, 20_000, nil}, |
| 41 | } { |
| 42 | t.Run(tc.name, func(t *testing.T) { |
| 43 | shape := requestCalibrationShape{requestChars: tc.chars} |
| 44 | if got := promptTruncationCeiling(tc.promptTokens, shape, tc.cal); got != 0 { |
| 45 | t.Fatalf("ceiling = %d, want 0 (no truncation)", got) |
| 46 | } |
| 47 | }) |
| 48 | } |
| 49 | } |
| 50 | |
| 51 | // The loop this closes: a truncated observation must never become the ratio |
| 52 | // every later estimate is built from. Refusing it is safe on first sight, so it |
| 53 | // does not wait for corroboration. |
| 54 | func TestTruncatedObservationDoesNotBecomeTheCalibration(t *testing.T) { |
| 55 | a := &Agent{} |
| 56 | shape := requestCalibrationShape{requestChars: 54_000} |
| 57 | a.setPromptTokenCalibration(2051, shape) |
| 58 | if cal := a.sess.output.promptCalibration.Load(); cal != nil { |
| 59 | t.Fatalf("truncated observation was stored as calibration: %+v", cal) |
| 60 | } |
| 61 | if _, ok := a.calibratedPromptTokens(shape); ok { |
| 62 | t.Fatal("estimates are calibrated from a truncated prompt") |
| 63 | } |
| 64 | } |
| 65 | |
| 66 | // Detection must never clamp the window: admission would then reject every |
| 67 | // prompt that does not fit, stopping a session that today merely degrades. |
| 68 | func TestTruncationNeverClampsTheWindow(t *testing.T) { |
| 69 | a := &Agent{} |
| 70 | a.setPromptTokenCalibration(2051, requestCalibrationShape{requestChars: 54_000}) |
| 71 | a.setPromptTokenCalibration(2048, requestCalibrationShape{requestChars: 90_000}) |
| 72 | if learned := a.sess.output.learned.Load(); learned != nil && learned.windowTokens > 0 { |
| 73 | t.Fatalf("detection clamped the window to %d", learned.windowTokens) |
| 74 | } |
| 75 | } |
| 76 | |
| 77 | // The measured session truncates on its second turn, so the user must be told |
| 78 | // on that turn rather than after another round trip. |
| 79 | func TestTheFirstTruncatedTurnIsTheOneThatWarns(t *testing.T) { |
| 80 | a := &Agent{} |
| 81 | a.setPromptTokenCalibration(2051, requestCalibrationShape{requestChars: 54_000}) |
| 82 | tr := a.sess.output.truncation.Load() |
| 83 | if tr == nil || !tr.notified || tr.promptCeiling != 2051 { |
| 84 | t.Fatalf("truncation state = %+v, want notified ceiling 2051", tr) |
| 85 | } |
| 86 | } |
| 87 | |
| 88 | // A provider reporting meaningless usage is not a small window. Stub and broken |
| 89 | // providers report token counts no runtime could be serving. |
| 90 | func TestImplausiblySmallUsageIsNotTreatedAsACeiling(t *testing.T) { |
| 91 | shape := requestCalibrationShape{requestChars: 200_000} |
| 92 | if got := promptTruncationCeiling(10, shape, nil); got != 0 { |
| 93 | t.Fatalf("ceiling = %d, want 0: 10 tokens is not a context window", got) |
| 94 | } |
| 95 | } |
| 96 | |
| 97 | func TestHonestObservationStillCalibrates(t *testing.T) { |
| 98 | a := &Agent{} |
| 99 | shape := requestCalibrationShape{requestChars: 20_000} |
| 100 | a.setPromptTokenCalibration(5_200, shape) |
| 101 | cal := a.sess.output.promptCalibration.Load() |
| 102 | if cal == nil || cal.promptTokens != 5_200 { |
| 103 | t.Fatalf("honest observation was rejected: %+v", cal) |
| 104 | } |
| 105 | } |
| 106 | |
| 107 | // A calibration that is itself implausible must not suppress detection: the |
| 108 | // absolute floor still applies underneath it. |
| 109 | func TestAnImplausibleCalibrationDoesNotSuppressDetection(t *testing.T) { |
| 110 | cal := &promptTokenCalibration{promptTokens: 400, requestChars: 100_000} // density 0.004 |
| 111 | shape := requestCalibrationShape{requestChars: 54_000} |
| 112 | if got := promptTruncationCeiling(2051, shape, cal); got != 2051 { |
| 113 | t.Fatalf("ceiling = %d, want 2051", got) |
| 114 | } |
| 115 | } |
| 116 | |
| 117 | // One warning per session, not one per racing turn. |
| 118 | func TestConcurrentTurnsWarnOnlyOnce(t *testing.T) { |
| 119 | a := &Agent{} |
| 120 | shape := requestCalibrationShape{requestChars: 54_000} |
| 121 | var wg sync.WaitGroup |
| 122 | for range 8 { |
| 123 | wg.Go(func() { a.setPromptTokenCalibration(2051, shape) }) |
| 124 | } |
| 125 | wg.Wait() |
| 126 | tr := a.sess.output.truncation.Load() |
| 127 | if tr == nil || !tr.notified { |
| 128 | t.Fatalf("truncation state = %+v, want notified", tr) |
| 129 | } |
| 130 | } |
| 131 |